wang2226's picture
Beyond Tokens decoding playground: contrastive, guided and parallel decoding
371d90c verified
Raw History Blame Contribute Delete
49.4 kB
[
{
"section": "Survey",
"title": "Comparison of Diverse Decoding Methods from Conditional Language Models",
"authors": "Daphne Ippolito, Reno Kriz, João Sedoc, Maria Kustikova, Chris Callison-Burch",
"abbr": "",
"venue": "ACL2019",
"model": "PLM",
"pdf": "https://aclanthology.org/P19-1365.pdf",
"code": "",
"year": 2019,
"group": "survey"
},
{
"section": "Survey",
"title": "On Decoding Strategies for Neural Text Generators",
"authors": "Gian Wiher, Clara Meister, Ryan Cotterell",
"abbr": "",
"venue": "TACL2022",
"model": "PLM",
"pdf": "https://aclanthology.org/2022.tacl-1.58.pdf",
"code": "",
"year": 2022,
"group": "survey"
},
{
"section": "Survey",
"title": "Unlocking Efficiency in Large Language Model Inference: A Comprehensive Survey of Speculative Decoding",
"authors": "Heming Xia, Zhe Yang, Qingxiu Dong, Peiyi Wang, Yongqi Li, Tao Ge, Tianyu Liu, Wenjie Li, Zhifang Sui",
"abbr": "",
"venue": "ACL2024",
"model": "LLM",
"pdf": "https://aclanthology.org/2024.findings-acl.456.pdf",
"code": "",
"year": 2024,
"group": "survey"
},
{
"section": "Survey",
"title": "From Decoding to Meta-Generation: Inference-time Algorithms for Large Language Models",
"authors": "Sean Welleck, Amanda Bertsch, Matthew Finlayson, Hailey Schoelkopf, Alex Xie, Graham Neubig, Ilia Kulikov, Zaid Harchaoui",
"abbr": "",
"venue": "TMLR",
"model": "LLM",
"pdf": "https://arxiv.org/pdf/2406.16838",
"code": "",
"year": 2024,
"group": "survey"
},
{
"section": "Survey",
"title": "Controllable Text Generation for Large Language Models: A Survey",
"authors": "Xun Liang, Hanyu Wang, Yezhaohui Wang, Shichao Song, Jiawei Yang, Simin Niu, Jie Hu, Dan Liu, Shunyu Yao, Feiyu Xiong, Zhiyu Li",
"abbr": "",
"venue": "",
"model": "LLM",
"pdf": "https://arxiv.org/pdf/2408.12599",
"code": "",
"year": 2024,
"group": "survey"
},
{
"section": "Paradigms › Contrastive",
"title": "DExperts: Decoding-Time Controlled Text Generation with Experts and Anti-Experts",
"authors": "Alisa Liu, Maarten Sap, Ximing Lu, Swabha Swayamdipta, Chandra Bhagavatula, Noah A. Smith, Yejin Choi",
"abbr": "DExperts",
"venue": "ACL2021",
"model": "PLM",
"pdf": "https://aclanthology.org/2023.findings-emnlp.257.pdf",
"code": "https://github.com/alisawuffles/DExperts",
"year": 2021,
"group": "contrastive"
},
{
"section": "Paradigms › Contrastive",
"title": "Contrastive Decoding: Open-ended Text Generation as Optimization",
"authors": "Xiang Lisa Li, Ari Holtzman, Daniel Fried, Percy Liang, Jason Eisner, Tatsunori Hashimoto, Luke Zettlemoyer, Mike Lewis",
"abbr": "CD",
"venue": "ACL2023",
"model": "PLM",
"pdf": "https://aclanthology.org/2023.acl-long.687.pdf",
"code": "https://github.com/XiangLi1999/ContrastiveDecoding",
"year": 2023,
"group": "contrastive"
},
{
"section": "Paradigms › Contrastive",
"title": "Trusting Your Evidence: Hallucinate Less with Context-aware Decoding",
"authors": "Weijia Shi, Xiaochuang Han, Mike Lewis, Yulia Tsvetkov, Luke Zettlemoyer, Wen-tau Yih",
"abbr": "CAD",
"venue": "NACL2024",
"model": "LLM",
"pdf": "https://aclanthology.org/2024.naacl-short.69.pdf",
"code": "https://github.com/xhan77/context-aware-decoding",
"year": 2024,
"group": "contrastive"
},
{
"section": "Paradigms › Contrastive",
"title": "Speculative Contrastive Decoding",
"authors": "Hongyi Yuan, Keming Lu, Fei Huang, Zheng Yuan, Chang Zhou",
"abbr": "SCD",
"venue": "ACL2024",
"model": "LLM",
"pdf": "https://aclanthology.org/2024.acl-short.5.pdf",
"code": "",
"year": 2024,
"group": "contrastive"
},
{
"section": "Paradigms › Contrastive",
"title": "DoLa: Decoding by Contrasting Layers Improves Factuality in Large Language Models",
"authors": "Yung-Sung Chuang, Yujia Xie, Hongyin Luo, Yoon Kim, James Glass, Pengcheng He",
"abbr": "DoLa",
"venue": "ICLR2024",
"model": "LLM",
"pdf": "https://arxiv.org/pdf/2309.03883",
"code": "https://github.com/voidism/DoLa",
"year": 2024,
"group": "contrastive"
},
{
"section": "Paradigms › Contrastive",
"title": "Mitigating Object Hallucinations in Large Vision-Language Models through Visual Contrastive Decoding",
"authors": "Sicong Leng, Hang Zhang, Guanzheng Chen, Xin Li, Shijian Lu, Chunyan Miao, Lidong Bing",
"abbr": "VCD",
"venue": "CVPR2024",
"model": "LVLM",
"pdf": "https://arxiv.org/pdf/2311.16922",
"code": "https://github.com/DAMO-NLP-SG/VCD",
"year": 2024,
"group": "contrastive"
},
{
"section": "Paradigms › Contrastive",
"title": "ROSE Doesn't Do That: Boosting the Safety of Instruction-Tuned Large Language Models with Reverse Prompt Contrastive Decoding",
"authors": "Qihuang Zhong, Liang Ding, Juhua Liu, Bo Du, Dacheng Tao",
"abbr": "ROSE",
"venue": "ACL2024-Findings",
"model": "LLM",
"pdf": "https://aclanthology.org/2024.findings-acl.814.pdf",
"code": "https://github.com/WHU-ZQH/ROSE",
"year": 2024,
"group": "contrastive"
},
{
"section": "Paradigms › Contrastive",
"title": "Enhancing Contextual Understanding in Large Language Models through Contrastive Decoding",
"authors": "Zheng Zhao, Emilio Monti, Jens Lehmann, Haytham Assem",
"abbr": "",
"venue": "NAACL2024",
"model": "LLM",
"pdf": "https://aclanthology.org/2024.naacl-long.237.pdf",
"code": "",
"year": 2024,
"group": "contrastive"
},
{
"section": "Paradigms › Contrastive",
"title": "Entropy-Based Decoding for Retrieval-Augmented Large Language Models",
"authors": "Zexuan Qiu, Zijing Ou, Bin Wu, Jingjing Li, Aiwei Liu, Irwin King",
"abbr": "CLeHe",
"venue": "",
"model": "LLM",
"pdf": "https://arxiv.org/pdf/2406.17519",
"code": "",
"year": 2024,
"group": "contrastive"
},
{
"section": "Paradigms › Contrastive",
"title": "Adaptive Contrastive Decoding in Retrieval-Augmented Generation for Handling Noisy Contexts",
"authors": "Youna Kim, Hyuhng Joon Kim, Cheonbok Park, Choonghyun Park, Hyunsoo Cho, Junyeob Kim, Kang Min Yoo, Sang-goo Lee, Taeuk Kim",
"abbr": "ACD",
"venue": "EMNLP2024-Findings",
"model": "LLM",
"pdf": "https://aclanthology.org/2024.findings-emnlp.136.pdf",
"code": "",
"year": 2024,
"group": "contrastive"
},
{
"section": "Paradigms › Contrastive",
"title": "Unchosen Experts Can Contribute Too: Unleashing MoE Models' Power by Self-Contrast",
"authors": "Chufan Shi, Cheng Yang, Xinyu Zhu, Jiahao Wang, Taiqiang Wu, Siheng Li, Deng Cai, Yujiu Yang, Yu Meng",
"abbr": "SCMoE",
"venue": "",
"model": "LLM",
"pdf": "https://arxiv.org/pdf/2405.14507v1",
"code": "https://github.com/DavidFanzz/SCMoE",
"year": 2024,
"group": "contrastive"
},
{
"section": "Paradigms › Contrastive",
"title": "Entropy Guided Extrapolative Decoding to Improve Factuality in Large Language Models",
"authors": "Souvik Das, Lifeng Jin, Linfeng Song, Haitao Mi, Baolin Peng, Dong Yu",
"abbr": "",
"venue": "",
"model": "LLM",
"pdf": "https://arxiv.org/pdf/2404.09338",
"code": "https://github.com/souvikdgp16/extrapolative_decoding",
"year": 2024,
"group": "contrastive"
},
{
"section": "Paradigms › Contrastive",
"title": "Mitigating Hallucinations in Large Vision-Language Models with Instruction Contrastive Decoding",
"authors": "Xintong Wang, Jingheng Pan, Liang Ding, Chris Biemann",
"abbr": "",
"venue": "ACL2024-Findings",
"model": "LVLM",
"pdf": "https://aclanthology.org/2024.findings-acl.937.pdf",
"code": "",
"year": 2024,
"group": "contrastive"
},
{
"section": "Paradigms › Contrastive",
"title": "IBD: Alleviating Hallucinations in Large Vision-Language Models via Image-Biased Decoding",
"authors": "Lanyun Zhu, Deyi Ji, Tianrun Chen, Peng Xu, Jieping Ye, Jun Liu",
"abbr": "IBD",
"venue": "",
"model": "LVLM",
"pdf": "https://arxiv.org/pdf/2402.18476",
"code": "",
"year": 2024,
"group": "contrastive"
},
{
"section": "Paradigms › Contrastive",
"title": "VACoDe: Visual Augmented Contrastive Decoding",
"authors": "Sihyeon Kim, Boryeong Cho, Sangmin Bae, Sumyeong Ahn, Se-Young Yun",
"abbr": "VACoDe",
"venue": "",
"model": "LVLM",
"pdf": "https://arxiv.org/pdf/2408.05337",
"code": "",
"year": 2024,
"group": "contrastive"
},
{
"section": "Paradigms › Contrastive",
"title": "VaLiD: Mitigating the Hallucination of Large Vision Language Models by Visual Layer Fusion Contrastive Decoding",
"authors": "Jiaqi Wang, Yifei Gao, Jitao Sang",
"abbr": "VaLiD",
"venue": "",
"model": "LVLM",
"pdf": "https://arxiv.org/pdf/2411.15839",
"code": "",
"year": 2024,
"group": "contrastive"
},
{
"section": "Paradigms › Contrastive",
"title": "Mitigating Hallucinations in Large Vision-Language Models (LVLMs) via Language-Contrastive Decoding (LCD)",
"authors": "Avshalom Manevich, Reut Tsarfaty",
"abbr": "LCD",
"venue": "ACL2024-Findings",
"model": "LVLM",
"pdf": "https://aclanthology.org/2024.findings-acl.359.pdf",
"code": "",
"year": 2024,
"group": "contrastive"
},
{
"section": "Paradigms › Contrastive",
"title": "Steering Multimodal Large Language Models Decoding for Context-Aware Safety",
"authors": "Zheyuan Liu, Zhangchen Xu, Guangyao Dou, Xiangchi Yuan, Zhaoxuan Tan, Radha Poovendran, Meng Jiang",
"abbr": "SafeCoDe",
"venue": "EMNLP2026",
"model": "LVLM",
"pdf": "https://arxiv.org/pdf/2509.19212",
"code": "",
"year": 2026,
"group": "contrastive"
},
{
"section": "Paradigms › Guided",
"title": "NeuroLogic Decoding: (Un)supervised Neural Text Generation with Predicate Logic Constraints",
"authors": "Ximing Lu, Peter West, Rowan Zellers, Ronan Le Bras, Chandra Bhagavatula, Yejin Choi",
"abbr": "",
"venue": "NAACL2021",
"model": "PLM",
"pdf": "https://aclanthology.org/2021.naacl-main.339.pdf",
"code": "",
"year": 2021,
"group": "guided"
},
{
"section": "Paradigms › Guided",
"title": "FUDGE: Controlled Text Generation With Future Discriminators",
"authors": "Kevin Yang, Dan Klein",
"abbr": "FUDGE",
"venue": "NAACL2021",
"model": "PLM",
"pdf": "https://aclanthology.org/2021.naacl-main.276.pdf",
"code": "https://github.com/yangkevin2/naacl-2021-fudge-controlled-generation",
"year": 2021,
"group": "guided"
},
{
"section": "Paradigms › Guided",
"title": "NeuroLogic A*esque Decoding: Constrained Text Generation with Lookahead Heuristics",
"authors": "Ximing Lu, Sean Welleck, Peter West, Liwei Jiang, Jungo Kasai, Daniel Khashabi, Ronan Le Bras, Lianhui Qin, Youngjae Yu, Rowan Zellers, Noah A. Smith, Yejin Choi",
"abbr": "",
"venue": "NAACL2022",
"model": "PLM",
"pdf": "https://aclanthology.org/2022.naacl-main.57.pdf",
"code": "",
"year": 2022,
"group": "guided"
},
{
"section": "Paradigms › Guided",
"title": "Critic-Guided Decoding for Controlled Text Generation",
"authors": "Minbeom Kim, Hwanhee Lee, Kang Min Yoo, Joonsuk Park, Hwaran Lee, Kyomin Jung",
"abbr": "CriticControl",
"venue": "ACL2023",
"model": "PLM",
"pdf": "https://aclanthology.org/2023.findings-acl.281.pdf",
"code": "https://github.com/minbeomkim/CriticControl",
"year": 2023,
"group": "guided"
},
{
"section": "Paradigms › Guided",
"title": "NaturalProver: Grounded Mathematical Proof Generation with Language Models",
"authors": "Sean Welleck, Jiacheng Liu, Ximing Lu, Hannaneh Hajishirzi, Yejin Choi",
"abbr": "NaturalProver",
"venue": "NeurIPS2022",
"model": "PLM",
"pdf": "https://arxiv.org/pdf/2205.12910",
"code": "https://github.com/wellecks/naturalprover",
"year": 2022,
"group": "guided"
},
{
"section": "Paradigms › Guided",
"title": "MIL-Decoding: Detoxifying Language Models at Token-Level via Multiple Instance Learning",
"authors": "Xu Zhang, Xiaojun Wan",
"abbr": "MIL-Decoding",
"venue": "ACL2023",
"model": "LLM",
"pdf": "https://aclanthology.org/2023.acl-long.11.pdf",
"code": "",
"year": 2023,
"group": "guided"
},
{
"section": "Paradigms › Guided",
"title": "Reward-Augmented Decoding: Efficient Controlled Text Generation With a Unidirectional Reward Model",
"authors": "Haikang Deng, Colin Raffel",
"abbr": "RAD",
"venue": "EMNLP2023",
"model": "LLM",
"pdf": "https://aclanthology.org/2023.emnlp-main.721.pdf",
"code": "",
"year": 2023,
"group": "guided"
},
{
"section": "Paradigms › Guided",
"title": "Don't throw away your value model! Generating more preferable text with Value-Guided Monte-Carlo Tree Search decoding",
"authors": "Jiacheng Liu, Andrew Cohen, Ramakanth Pasunuru, Yejin Choi, Hannaneh Hajishirzi, Asli Celikyilmaz",
"abbr": "PPO-MCTS",
"venue": "",
"model": "LLM",
"pdf": "https://arxiv.org/pdf/2309.15028",
"code": "",
"year": 2023,
"group": "guided"
},
{
"section": "Paradigms › Guided",
"title": "Planning with Large Language Models for Code Generation",
"authors": "Shun Zhang, Zhenfang Chen, Yikang Shen, Mingyu Ding, Joshua B. Tenenbaum, Chuang Gan",
"abbr": "PG-TD",
"venue": "ICLR2023",
"model": "LLM",
"pdf": "https://arxiv.org/pdf/2303.05510",
"code": "https://codeaimcts.github.io/",
"year": 2023,
"group": "guided"
},
{
"section": "Paradigms › Guided",
"title": "Seeing is Believing: Mitigating Hallucination in Large Vision-Language Models via CLIP-Guided Decoding",
"authors": "Ailin Deng, Zhirui Chen, Bryan Hooi",
"abbr": "CGD",
"venue": "",
"model": "LVLM",
"pdf": "https://arxiv.org/pdf/2402.15300",
"code": "https://github.com/d-ailin/CLIP-Guided-Decoding",
"year": 2024,
"group": "guided"
},
{
"section": "Paradigms › Guided",
"title": "Deductive Beam Search: Decoding Deducible Rationale for Chain-of-Thought Reasoning",
"authors": "Tinghui Zhu, Kai Zhang, Jian Xie, Yu Su",
"abbr": "DBS",
"venue": "COLM2024",
"model": "LLM",
"pdf": "https://arxiv.org/pdf/2401.17686",
"code": "https://github.com/OSU-NLP-Group/Deductive-Beam-Search",
"year": 2024,
"group": "guided"
},
{
"section": "Paradigms › Guided",
"title": "A Data-Driven Guided Decoding Mechanism for Diagnostic Captioning",
"authors": "Panagiotis Kaliosis, John Pavlopoulos, Foivos Charalampakos, Georgios Moschovis, Ion Androutsopoulos",
"abbr": "DMMCS",
"venue": "ACL2024-Findings",
"model": "LVLM",
"pdf": "https://aclanthology.org/2024.findings-acl.444.pdf",
"code": "https://github.com/nlpaueb/dmmcs",
"year": 2024,
"group": "guided"
},
{
"section": "Paradigms › Guided",
"title": "Mitigating Hallucinations in Large Vision-Language Models via Summary-Guided Decoding",
"authors": "Kyungmin Min, Minbeom Kim, Kang-il Lee, Dongryeol Lee, Kyomin Jung",
"abbr": "SGD",
"venue": "",
"model": "LVLM",
"pdf": "https://arxiv.org/pdf/2410.13321",
"code": "",
"year": 2024,
"group": "guided"
},
{
"section": "Paradigms › Guided",
"title": "Self-Introspective Decoding: Alleviating Hallucinations for Large Vision-Language Models",
"authors": "Fushuo Huo, Wenchao Xu, Zhong Zhang, Haozhao Wang, Zhicheng Chen, Peilin Zhao",
"abbr": "SID",
"venue": "",
"model": "LVLM",
"pdf": "https://arxiv.org/pdf/2408.02032",
"code": "https://github.com/huofushuo/SID",
"year": 2024,
"group": "guided"
},
{
"section": "Paradigms › Guided",
"title": "Alphazero-like Tree-Search can Guide Large Language Model Decoding and Training",
"authors": "Xidong Feng, Ziyu Wan, Muning Wen, Stephen Marcus McAleer, Ying Wen, Weinan Zhang, Jun Wang",
"abbr": "TS-LLM",
"venue": "",
"model": "LLM",
"pdf": "https://arxiv.org/pdf/2309.17179",
"code": "https://github.com/waterhorse1/LLM_Tree_Search",
"year": 2023,
"group": "guided"
},
{
"section": "Paradigms › Guided",
"title": "From Uncertainty to Trust: Enhancing Reliability in Vision-Language Models with Uncertainty-Guided Dropout Decoding",
"authors": "Yixiong Fang, Ziran Yang, Zhaorun Chen, Zhuokai Zhao, Jiawei Zhou",
"abbr": "Dropout",
"venue": "",
"model": "LVLM",
"pdf": "https://arxiv.org/pdf/2412.06474",
"code": "https://github.com/kigb/DropoutDecoding",
"year": 2024,
"group": "guided"
},
{
"section": "Paradigms › Guided",
"title": "Monitor-Guided Decoding of Code LMs with Static Analysis of Repository Context",
"authors": "Lakshya A Agrawal, Aditya Kanade, Navin Goyal, Shuvendu K. Lahiri, Sriram K. Rajamani",
"abbr": "MGD",
"venue": "NeurIPS2023",
"model": "LLM",
"pdf": "https://proceedings.neurips.cc/paper_files/paper/2023/file/662b1774ba8845fc1fa3d1fc0177ceeb-Paper-Conference.pdf",
"code": "https://github.com/microsoft/monitors4codegen",
"year": 2023,
"group": "guided"
},
{
"section": "Paradigms › Guided",
"title": "SafeDecoding: Defending against Jailbreak Attacks via Safety-Aware Decoding",
"authors": "Zhangchen Xu, Fengqing Jiang, Luyao Niu, Jinyuan Jia, Bill Yuchen Lin, Radha Poovendran",
"abbr": "SafeDecoding",
"venue": "ACL2024",
"model": "LLM",
"pdf": "https://aclanthology.org/2024.acl-long.303.pdf",
"code": "https://github.com/uw-nsl/SafeDecoding",
"year": 2024,
"group": "guided"
},
{
"section": "Paradigms › Guided",
"title": "Guiding LLMs The Right Way: Fast, Non-Invasive Constrained Generation",
"authors": "Luca Beurer-Kellner, Marc Fischer, Martin Vechev",
"abbr": "DOMINO",
"venue": "ICML2024",
"model": "LLM",
"pdf": "https://arxiv.org/pdf/2403.06988v1",
"code": "https://github.com/eth-sri/domino",
"year": 2024,
"group": "guided"
},
{
"section": "Paradigms › Guided",
"title": "Odysseus Navigates the Sirens' Song: Dynamic Focus Decoding for Factual and Diverse Open-Ended Text Generation",
"authors": "Wen Luo, Feifan Song, Wei Li, Guangyue Peng, Shaohang Wei, Houfeng Wang",
"abbr": "DFD",
"venue": "",
"model": "LLM",
"pdf": "https://arxiv.org/pdf/2503.08057",
"code": "",
"year": 2025,
"group": "guided"
},
{
"section": "Paradigms › Guided",
"title": "Attention Reallocation: Towards Zero-cost and Controllable Hallucination Mitigation of MLLMs",
"authors": "Chongjun Tu, Peng Ye, Dongzhan Zhou, Lei Bai, Gang Yu, Tao Chen, Wanli Ouyang",
"abbr": "AttnReal",
"venue": "",
"model": "LVLM",
"pdf": "https://arxiv.org/pdf/2503.08342",
"code": "",
"year": 2025,
"group": "guided"
},
{
"section": "Paradigms › Guided",
"title": "Learning to Align: Addressing Character Frequency Distribution Shifts in Handwritten Text Recognition",
"authors": "Panagiotis Kaliosis, John Pavlopoulos",
"abbr": "FADA",
"venue": "EMNLP2025",
"model": "LLM",
"pdf": "https://aclanthology.org/2025.findings-emnlp.1014.pdf",
"code": "",
"year": 2025,
"group": "guided"
},
{
"section": "Paradigms › Guided",
"title": "PrefixNLI: Detecting Factual Inconsistencies as Soon as They Arise",
"authors": "Sapir Harary, Eran Hirsch, Aviv Slobodkin, David Wan, Mohit Bansal, Ido Dagan",
"abbr": "PrefixNLI",
"venue": "",
"model": "LLM",
"pdf": "https://arxiv.org/abs/2511.01359",
"code": "",
"year": 2025,
"group": "guided"
},
{
"section": "Paradigms › Parallel",
"title": "Blockwise Parallel Decoding for Deep Autoregressive Models",
"authors": "Mitchell Stern, Noam Shazeer, Jakob Uszkoreit",
"abbr": "Blockwise",
"venue": "NeurIPS2018",
"model": "PLM",
"pdf": "https://proceedings.neurips.cc/paper_files/paper/2018/file/c4127b9194fe8562c64dc0f5bf2c93bc-Paper.pdf",
"code": "",
"year": 2018,
"group": "parallel"
},
{
"section": "Paradigms › Parallel",
"title": "Speculative Decoding: Exploiting Speculative Execution for Accelerating Seq2seq Generation",
"authors": "Heming Xia, Tao Ge, Peiyi Wang, Si-Qing Chen, Furu Wei, Zhifang Sui",
"abbr": "SpecDec",
"venue": "EMNLP2023-Findings",
"model": "PLM",
"pdf": "https://aclanthology.org/2023.findings-emnlp.257.pdf",
"code": "https://github.com/hemingkx/SpecDec",
"year": 2023,
"group": "parallel"
},
{
"section": "Paradigms › Parallel",
"title": "Accelerating Transformer Inference for Translation via Parallel Decoding",
"authors": "Andrea Santilli, Silvio Severino, Emilian Postolache, Valentino Maiorca, Michele Mancusi, Riccardo Marin, Emanuele Rodolà",
"abbr": "",
"venue": "EMNLP2023-Findings",
"model": "PLM",
"pdf": "https://aclanthology.org/2023.acl-long.689.pdf",
"code": "https://github.com/teelinsan/parallel-decoding",
"year": 2023,
"group": "parallel"
},
{
"section": "Paradigms › Parallel",
"title": "Draft& Verify: Lossless Large Language Model Acceleration via Self-Speculative Decoding",
"authors": "Jun Zhang, Jue Wang, Huan Li, Lidan Shou, Ke Chen, Gang Chen, Sharad Mehrotra",
"abbr": "Self-Speculative",
"venue": "ACL2024",
"model": "LLM",
"pdf": "https://aclanthology.org/2024.acl-long.607.pdf",
"code": "https://github.com/dilab-zju/self-speculative-decoding",
"year": 2024,
"group": "parallel"
},
{
"section": "Paradigms › Parallel",
"title": "Fast Inference from Transformers via Speculative Decoding",
"authors": "Yaniv Leviathan, Matan Kalman, Yossi Matias",
"abbr": "",
"venue": "ICML2023",
"model": "LLM",
"pdf": "https://arxiv.org/pdf/2211.17192",
"code": "",
"year": 2023,
"group": "parallel"
},
{
"section": "Paradigms › Parallel",
"title": "Accelerating Large Language Model Decoding with Speculative Sampling",
"authors": "Charlie Chen, Sebastian Borgeaud, Geoffrey Irving, Jean-Baptiste Lespiau, Laurent Sifre, John Jumper",
"abbr": "",
"venue": "",
"model": "LLM",
"pdf": "https://arxiv.org/pdf/2302.01318",
"code": "",
"year": 2023,
"group": "parallel"
},
{
"section": "Paradigms › Parallel",
"title": "DistillSpec: Improving Speculative Decoding via Knowledge Distillation",
"authors": "Yongchao Zhou, Kaifeng Lyu, Ankit Singh Rawat, Aditya Krishna Menon, Afshin Rostamizadeh, Sanjiv Kumar, Jean-François Kagy, Rishabh Agarwal",
"abbr": "DistillSpec",
"venue": "ICLR2024",
"model": "LLM",
"pdf": "https://arxiv.org/pdf/2310.08461",
"code": "",
"year": 2024,
"group": "parallel"
},
{
"section": "Paradigms › Parallel",
"title": "SpecInfer: Accelerating Generative Large Language Model Serving with Tree-based Speculative Inference and Verification",
"authors": "Xupeng Miao, Gabriele Oliaro, Zhihao Zhang, Xinhao Cheng, Zeyu Wang, Zhengxin Zhang, Rae Ying Yee Wong, Alan Zhu, Lijie Yang, Xiaoxiang Shi, Chunan Shi, Zhuoming Chen, Daiyaan Arfeen, Reyna Abhyankar, Zhihao Jia",
"abbr": "Self-SpecInfer",
"venue": "ASPLOS2024",
"model": "LLM",
"pdf": "https://arxiv.org/pdf/2305.09781",
"code": "https://github.com/flexflow/flexflow-train",
"year": 2024,
"group": "parallel"
},
{
"section": "Paradigms › Parallel",
"title": "Online Speculative Decoding",
"authors": "Xiaoxuan Liu, Lanxiang Hu, Peter Bailis, Alvin Cheung, Zhijie Deng, Ion Stoica, Hao Zhang",
"abbr": "Online-Speculative",
"venue": "ICML2024",
"model": "LLM",
"pdf": "https://arxiv.org/pdf/2310.07177",
"code": "https://github.com/LiuXiaoxuanPKU/OSD",
"year": 2024,
"group": "parallel"
},
{
"section": "Paradigms › Parallel",
"title": "Speculative RAG: Enhancing Retrieval Augmented Generation through Drafting",
"authors": "Zilong Wang, Zifeng Wang, Long Le, Huaixiu Steven Zheng, Swaroop Mishra, Vincent Perot, Yuwei Zhang, Anush Mattapalli, Ankur Taly, Jingbo Shang, Chen-Yu Lee, Tomas Pfister",
"abbr": "SpeculativeRAG",
"venue": "",
"model": "LLM",
"pdf": "https://arxiv.org/pdf/2407.08223",
"code": "",
"year": 2024,
"group": "parallel"
},
{
"section": "Paradigms › Parallel",
"title": "Break the Sequential Dependency of LLM Inference Using Lookahead Decoding",
"authors": "Yichao Fu, Peter Bailis, Ion Stoica, Hao Zhang",
"abbr": "Lookahead",
"venue": "",
"model": "LLM",
"pdf": "https://arxiv.org/pdf/2402.02057",
"code": "https://github.com/hao-ai-lab/LookaheadDecoding",
"year": 2024,
"group": "parallel"
},
{
"section": "Paradigms › Parallel",
"title": "Medusa: Simple LLM Inference Acceleration Framework with Multiple Decoding Heads",
"authors": "Tianle Cai, Yuhong Li, Zhengyang Geng, Hongwu Peng, Jason D. Lee, Deming Chen, Tri Dao",
"abbr": "Medusa",
"venue": "ICML2024",
"model": "LLM",
"pdf": "https://arxiv.org/pdf/2401.10774",
"code": "https://github.com/FasterDecoding/Medusa",
"year": 2024,
"group": "parallel"
},
{
"section": "Paradigms › Parallel",
"title": "EAGLE: Speculative Sampling Requires Rethinking Feature Uncertainty",
"authors": "Yuhui Li, Fangyun Wei, Chao Zhang, Hongyang Zhang",
"abbr": "Eagle",
"venue": "ICML2024",
"model": "LLM",
"pdf": "https://arxiv.org/pdf/2401.15077",
"code": "https://github.com/SafeAILab/EAGLE",
"year": 2024,
"group": "parallel"
},
{
"section": "Paradigms › Parallel",
"title": "EAGLE-2: Faster Inference of Language Models with Dynamic Draft Trees",
"authors": "Yuhui Li, Fangyun Wei, Chao Zhang, Hongyang Zhang",
"abbr": "Eagle",
"venue": "EMNLP2024",
"model": "LLM",
"pdf": "https://arxiv.org/pdf/2406.16858",
"code": "https://github.com/SafeAILab/EAGLE",
"year": 2024,
"group": "parallel"
},
{
"section": "Paradigms › Parallel",
"title": "On Speculative Decoding for Multimodal Large Language Models",
"authors": "Mukul Gagrani, Raghavv Goel, Wonseok Jeon, Junyoung Park, Mingu Lee, Christopher Lott",
"abbr": "",
"venue": "",
"model": "LVLM",
"pdf": "https://arxiv.org/pdf/2404.08856",
"code": "",
"year": 2024,
"group": "parallel"
},
{
"section": "Paradigms › Parallel",
"title": "LANTERN: Accelerating Visual Autoregressive Models with Relaxed Speculative Decoding",
"authors": "Doohyuk Jang, Sihwan Park, June Yong Yang, Yeonsung Jung, Jihun Yun, Souvik Kundu, Sung-Yub Kim, Eunho Yang",
"abbr": "Lantern",
"venue": "",
"model": "LVLM",
"pdf": "https://arxiv.org/pdf/2410.03355v1",
"code": "",
"year": 2024,
"group": "parallel"
},
{
"section": "Paradigms › Parallel",
"title": "Accelerating Auto-regressive Text-to-Image Generation with Training-free Speculative Jacobi Decoding",
"authors": "Yao Teng, Han Shi, Xian Liu, Xuefei Ning, Guohao Dai, Yu Wang, Zhenguo Li, Xihui Liu",
"abbr": "SJD",
"venue": "",
"model": "LVLM",
"pdf": "https://arxiv.org/pdf/2410.01699",
"code": "",
"year": 2024,
"group": "parallel"
},
{
"section": "Paradigms › Parallel",
"title": "Superposed Decoding: Multiple Generations from a Single Autoregressive Inference Pass",
"authors": "Ethan Shen, Alan Fan, Sarah M. Pratt, Jae Sung Park, Matthew Wallingford, Sham M. Kakade, Ari Holtzman, Ranjay Krishna, Ali Farhadi, Aditya Kusupati",
"abbr": "SuperposedDecoding",
"venue": "NeurIPS2024",
"model": "LLM",
"pdf": "https://arxiv.org/pdf/2405.18400",
"code": "https://github.com/RAIVNLab/SuperposedDecoding",
"year": 2024,
"group": "parallel"
},
{
"section": "Paradigms › Parallel",
"title": "SWIFT: On-the-Fly Self-Speculative Decoding for LLM Inference Acceleration",
"authors": "Heming Xia, Yongqi Li, Jun Zhang, Cunxiao Du, Wenjie Li",
"abbr": "Swift",
"venue": "",
"model": "LLM",
"pdf": "https://arxiv.org/pdf/2410.06916",
"code": "https://github.com/hemingkx/SWIFT",
"year": 2024,
"group": "parallel"
},
{
"section": "Paradigms › Parallel",
"title": "Seesaw: High-throughput LLM Inference via Model Re-sharding",
"authors": "Qidong Su, Wei Zhao, Xin Li, Muralidhar Andoorveedu, Chenhao Jiang, Zhanda Zhu, Kevin Song, Christina Giannoula, Gennady Pekhimenko",
"abbr": "Seesaw",
"venue": "",
"model": "LLM",
"pdf": "https://arxiv.org/pdf/2503.06433",
"code": "",
"year": 2025,
"group": "parallel"
},
{
"section": "Paradigms › Parallel",
"title": "DIVERSED: Relaxed Speculative Decoding via Dynamic Ensemble Verification",
"authors": "Ziyi Wang, Siva Rajesh Kasa, Ankith M S, Santhosh Kumar Kasa, Jiaru Zou, Sumit Negi, Ruqi Zhang, Nan Jiang, Qifan Song",
"abbr": "DIVERSED",
"venue": "AISTATS2026",
"model": "LLM",
"pdf": "https://arxiv.org/abs/2604.07622",
"code": "https://github.com/comeusr/diversed",
"year": 2026,
"group": "parallel"
},
{
"section": "Applications › Improve Model Alignment › Mitigate Hallucination",
"title": "DeCoRe: Decoding by Contrasting Retrieval Heads to Mitigate Hallucinations",
"authors": "Aryo Pradipta Gema, Chen Jin, Ahmed Abdulaal, Tom Diethe, Philip Teare, Beatrice Alex, Pasquale Minervini, Amrutha Saseendran",
"abbr": "DeCoRe",
"venue": "",
"model": "LLM",
"pdf": "https://arxiv.org/pdf/2410.18860",
"code": "https://github.com/aryopg/DeCoRe",
"year": 2024,
"group": "application"
},
{
"section": "Applications › Improve Model Alignment › Mitigate Hallucination",
"title": "Improving Factuality in Large Language Models via Decoding-Time Hallucinatory and Truthful Comparators",
"authors": "Dingkang Yang, Dongling Xiao, Jinjie Wei, Mingcheng Li, Zhaoyu Chen, Ke Li, Lihua Zhang",
"abbr": "CDT",
"venue": "",
"model": "LLM",
"pdf": "https://arxiv.org/pdf/2408.12325v1",
"code": "",
"year": 2024,
"group": "application"
},
{
"section": "Applications › Improve Model Alignment › Mitigate Hallucination",
"title": "Delve into Visual Contrastive Decoding for Hallucination Mitigation of Large Vision-Language Models",
"authors": "Yi-Lun Lee, Yi-Hsuan Tsai, Wei-Chen Chiu",
"abbr": "",
"venue": "",
"model": "LVLM",
"pdf": "https://arxiv.org/pdf/2412.06775",
"code": "https://github.com/YiLunLee/VCD_Analysis",
"year": 2024,
"group": "application"
},
{
"section": "Applications › Improve Model Alignment › Mitigate Hallucination",
"title": "ConVis: Contrastive Decoding with Hallucination Visualization for Mitigating Hallucinations in Multimodal Large Language Models",
"authors": "Yeji Park, Deokyeong Lee, Junsuk Choe, Buru Chang",
"abbr": "ConVis",
"venue": "",
"model": "LVLM",
"pdf": "https://arxiv.org/pdf/2408.13906",
"code": "https://github.com/yejipark-m/ConVis",
"year": 2024,
"group": "application"
},
{
"section": "Applications › Improve Model Alignment › Mitigate Hallucination",
"title": "MLLM can see? Dynamic Correction Decoding for Hallucination Mitigation",
"authors": "Chenxi Wang, Xiang Chen, Ningyu Zhang, Bozhong Tian, Haoming Xu, Shumin Deng, Huajun Chen",
"abbr": "DeCo",
"venue": "",
"model": "LVLM",
"pdf": "https://arxiv.org/pdf/2410.11779",
"code": "https://github.com/zjunlp/DeCo",
"year": 2024,
"group": "application"
},
{
"section": "Applications › Improve Model Alignment › Mitigate Hallucination",
"title": "CATCH: Complementary Adaptive Token-level Contrastive Decoding to Mitigate Hallucinations in LVLMs",
"authors": "Zhehan Kan, Ce Zhang, Zihan Liao, Yapeng Tian, Wenming Yang, Junyuan Xiao, Xu Li, Dongmei Jiang, Yaowei Wang, Qingmin Liao",
"abbr": "ID",
"venue": "ICLR2025",
"model": "LLM",
"pdf": "https://arxiv.org/pdf/2410.01556",
"code": "",
"year": 2025,
"group": "application"
},
{
"section": "Applications › Improve Model Alignment › Mitigate Hallucination",
"title": "Integrative Decoding: Improve Factuality via Implicit Self-consistency",
"authors": "Yi Cheng, Xiao Liang, Yeyun Gong, Wen Xiao, Song Wang, Yuji Zhang, Wenjun Hou, Kaishuai Xu, Wenge Liu, Wenjie Li, Jian Jiao, Qi Chen, Peng Cheng, Wayne Xiong",
"abbr": "CATCH",
"venue": "",
"model": "LVLM",
"pdf": "https://arxiv.org/pdf/2410.01556",
"code": "https://github.com/YiCheng98/IntegrativeDecoding",
"year": 2024,
"group": "application"
},
{
"section": "Applications › Improve Model Alignment › Mitigate Hallucination",
"title": "Attention Hijackers: Detect and Disentangle Attention Hijacking in LVLMs for Hallucination Mitigation",
"authors": "Beitao Chen, Xinyu Lyu, Lianli Gao, Jingkuan Song, Heng Tao Shen",
"abbr": "AID",
"venue": "",
"model": "LVLM",
"pdf": "https://arxiv.org/pdf/2503.08216",
"code": "",
"year": 2025,
"group": "application"
},
{
"section": "Applications › Improve Model Alignment › Improve Safety",
"title": "SafeInfer: Context Adaptive Decoding Time Safety Alignment for Large Language Models",
"authors": "Somnath Banerjee, Sayan Layek, Soham Tripathy, Shanu Kumar, Animesh Mukherjee, Rima Hazra",
"abbr": "SafeInfer",
"venue": "AAAI2025",
"model": "LLM",
"pdf": "https://arxiv.org/pdf/2406.12274",
"code": "https://github.com/NeuralSentinel/SafeInfer",
"year": 2025,
"group": "application"
},
{
"section": "Applications › Improve Model Alignment › Improve Safety",
"title": "Adversarial Contrastive Decoding: Boosting Safety Alignment of Large Language Models via Opposite Prompt Optimization",
"authors": "Zhengyue Zhao, Xiaoyun Zhang, Kaidi Xu, Xing Hu, Rui Zhang, Zidong Du, Qi Guo, Yunji Chen",
"abbr": "ACD",
"venue": "",
"model": "LLM",
"pdf": "https://arxiv.org/pdf/2406.16743",
"code": "",
"year": 2024,
"group": "application"
},
{
"section": "Applications › Improve Model Alignment › Improve Safety",
"title": "Adversarial Contrastive Decoding: Boosting Safety Alignment of Large Language Models via Opposite Prompt Optimization",
"authors": "Zhengyue Zhao, Xiaoyun Zhang, Kaidi Xu, Xing Hu, Rui Zhang, Zidong Du, Qi Guo, Yunji Chen",
"abbr": "ACD",
"venue": "",
"model": "LLM",
"pdf": "https://arxiv.org/pdf/2406.16743",
"code": "",
"year": 2024,
"group": "application"
},
{
"section": "Applications › Improve Model Alignment › Improve Safety",
"title": "Root Defence Strategies: Ensuring Safety of LLM at the Decoding Level",
"authors": "Xinyi Zeng, Yuying Shang, Yutao Zhu, Jiawei Chen, Yu Tian",
"abbr": "",
"venue": "",
"model": "LLM",
"pdf": "https://arxiv.org/pdf/2410.06809",
"code": "",
"year": 2024,
"group": "application"
},
{
"section": "Applications › Improve Model Alignment › Improve Safety",
"title": "Probing the Safety Response Boundary of Large Language Models via Unsafe Decoding Path Generation",
"authors": "Haoyu Wang, Bingzhe Wu, Yatao Bian, Yongzhe Chang, Xueqian Wang, Peilin Zhao",
"abbr": "JVD",
"venue": "",
"model": "LLM",
"pdf": "https://arxiv.org/pdf/2408.10668v1",
"code": "",
"year": 2024,
"group": "application"
},
{
"section": "Applications › Improve Model Alignment › Improve Safety",
"title": "Parameter-Efficient Detoxification with Contrastive Decoding",
"authors": "Tong Niu, Caiming Xiong, Yingbo Zhou, Semih Yavuz",
"abbr": "DETOXIGEN",
"venue": "",
"model": "LLM",
"pdf": "https://aclanthology.org/2024.hucllm-1.3.pdf",
"code": "",
"year": 2024,
"group": "application"
},
{
"section": "Applications › Improve Model Alignment › Improve Safety",
"title": "Transfer Q Star: Principled Decoding for LLM Alignment",
"authors": "Souradip Chakraborty, Soumya Suvra Ghosal, Ming Yin, Dinesh Manocha, Mengdi Wang, Amrit Singh Bedi, Furong Huang",
"abbr": "",
"venue": "",
"model": "LLM",
"pdf": "https://arxiv.org/pdf/2405.20495",
"code": "",
"year": 2024,
"group": "application"
},
{
"section": "Applications › Improve Model Alignment › Improve Safety",
"title": "Decoding Matters: Addressing Amplification Bias and Homogeneity Issue for LLM-based Recommendation",
"authors": "Keqin Bao, Jizhi Zhang, Yang Zhang, Xinyue Huo, Chong Chen, Fuli Feng",
"abbr": "",
"venue": "EMNLP2024",
"model": "LLM",
"pdf": "https://aclanthology.org/2024.emnlp-main.589.pdf",
"code": "https://github.com/SAI990323/DecodingMatters",
"year": 2024,
"group": "application"
},
{
"section": "Applications › Improve Model Alignment › Improve Safety",
"title": "Privacy-Aware Decoding: Mitigating Privacy Leakage of Large Language Models in Retrieval-Augmented Generation",
"authors": "Haoran Wang, Xiongxiao Xu, Baixiang Huang, Kai Shu",
"abbr": "",
"venue": "KDD2026",
"model": "LLM",
"pdf": "https://dl.acm.org/doi/epdf/10.1145/3770855.3817665",
"code": "https://github.com/wang2226/PAD",
"year": 2026,
"group": "application"
},
{
"section": "Applications › Improve Model Alignment › Improve Reasoning",
"title": "Contrastive Decoding Improves Reasoning in Large Language Models",
"authors": "Sean O'Brien, Mike Lewis",
"abbr": "",
"venue": "",
"model": "LLM",
"pdf": "https://arxiv.org/pdf/2309.09117",
"code": "",
"year": 2023,
"group": "application"
},
{
"section": "Applications › Improve Model Alignment › Improve Reasoning",
"title": "Distillation Contrastive Decoding: Improving LLMs Reasoning with Contrastive Decoding and Distillation",
"authors": "Phuc Phan, Hieu Tran, Long Phan",
"abbr": "DCD",
"venue": "",
"model": "LLM",
"pdf": "https://arxiv.org/pdf/2402.14874",
"code": "https://github.com/pphuc25/distil-cd",
"year": 2024,
"group": "application"
},
{
"section": "Applications › Improve Model Alignment › Improve Reasoning",
"title": "Expediting and Elevating Large Language Model Reasoning via Hidden Chain-of-Thought Decoding",
"authors": "Tianqiao Liu, Zui Chen, Zitao Liu, Mi Tian, Weiqi Luo",
"abbr": "",
"venue": "",
"model": "LLM",
"pdf": "https://arxiv.org/pdf/2409.08561",
"code": "",
"year": 2024,
"group": "application"
},
{
"section": "Applications › Improve Model Alignment › Improve Reasoning",
"title": "SEED: Accelerating Reasoning Tree Construction via Scheduled Speculative Decoding",
"authors": "Zhenglin Wang, Jialong Wu, Yilong Lai, Congzhi Zhang, Deyu Zhou",
"abbr": "SEED",
"venue": "",
"model": "LLM",
"pdf": "https://arxiv.org/pdf/2406.18200",
"code": "https://github.com/Linking-ai/SEED",
"year": 2024,
"group": "application"
},
{
"section": "Applications › Improve Model Alignment › Improve Reasoning",
"title": "Chain-of-Thought Reasoning Without Prompting",
"authors": "Xuezhi Wang, Denny Zhou",
"abbr": "",
"venue": "",
"model": "LLM",
"pdf": "https://arxiv.org/pdf/2402.10200",
"code": "",
"year": 2024,
"group": "application"
},
{
"section": "Applications › Improve Model Alignment › Improve Reasoning",
"title": "Self-Para-Consistency: Improving Reasoning Tasks at Low Cost for Large Language Models",
"authors": "Wenqing Chen, Weicheng Wang, Zhixuan Chu, Kui Ren, Zibin Zheng, Zhichao Lu",
"abbr": "",
"venue": "ACL2024-Findings",
"model": "LLM",
"pdf": "https://aclanthology.org/2024.findings-acl.842.pdf",
"code": "https://github.com/Linking-ai/SEED",
"year": 2024,
"group": "application"
},
{
"section": "Applications › Improve Model Alignment › Improve Reasoning",
"title": "Self-Evaluation Guided Beam Search for Reasoning",
"authors": "Yuxi Xie, Kenji Kawaguchi, Yiran Zhao, Xu Zhao, Min-Yen Kan, Junxian He, Qizhe Xie",
"abbr": "",
"venue": "NeurIPS2023",
"model": "LLM",
"pdf": "https://arxiv.org/pdf/2305.00633",
"code": "https://guideddecoding.github.io/",
"year": 2023,
"group": "application"
},
{
"section": "Applications › Improve Model Alignment › Improve Reasoning",
"title": "Learning to Decode Collaboratively with Multiple Language Models",
"authors": "Zejiang Shen, Hunter Lang, Bailin Wang, Yoon Kim, David Sontag",
"abbr": "Co-LLM",
"venue": "ACL2024",
"model": "LLM",
"pdf": "https://aclanthology.org/2024.acl-long.701.pdf",
"code": "https://github.com/clinicalml/co-llm",
"year": 2024,
"group": "application"
},
{
"section": "Applications › Improve Model Alignment › Improve Reasoning",
"title": "The Era of Semantic Decoding",
"authors": "Maxime Peyrard, Martin Josifoski, Robert West",
"abbr": "Semantic-Decoding",
"venue": "",
"model": "LLM",
"pdf": "https://arxiv.org/pdf/2403.14562",
"code": "",
"year": 2024,
"group": "application"
},
{
"section": "Applications › Improve Generation Tasks › RAG",
"title": "REST: Retrieval-Based Speculative Decoding",
"authors": "Zhenyu He, Zexuan Zhong, Tianle Cai, Jason Lee, Di He",
"abbr": "REST",
"venue": "NAACL2024",
"model": "LLM",
"pdf": "https://aclanthology.org/2024.naacl-long.88.pdf",
"code": "https://github.com/FasterDecoding/REST",
"year": 2024,
"group": "application"
},
{
"section": "Applications › Improve Generation Tasks › RAG",
"title": "Nonparametric Decoding for Generative Retrieval",
"authors": "Hyunji Lee, JaeYoung Kim, Hoyeon Chang, Hanseok Oh, Sohee Yang, Vladimir Karpukhin, Yi Lu, Minjoon Seo",
"abbr": "",
"venue": "ACL2023-Findings",
"model": "LLM",
"pdf": "https://aclanthology.org/2023.findings-acl.801.pdf",
"code": "https://github.com/amy-hyunji/Contextualized-Generative-Retrieval",
"year": 2023,
"group": "application"
},
{
"section": "Applications › Improve Generation Tasks › RAG",
"title": "Planning Ahead in Generative Retrieval: Guiding Autoregressive Generation through Simultaneous Decoding",
"authors": "Hansi Zeng, Chen Luo, Hamed Zamani",
"abbr": "PAG",
"venue": "SIGIR2024",
"model": "LLM",
"pdf": "https://dl.acm.org/doi/pdf/10.1145/3626772.3657746",
"code": "https://github.com/HansiZeng/PAG",
"year": 2024,
"group": "application"
},
{
"section": "Applications › Improve Generation Tasks › Code Generation",
"title": "DOCE: Finding the Sweet Spot for Execution-Based Code Generation",
"authors": "Haau-Sing Li, Patrick Fernandes, Iryna Gurevych, André F.T. Martins",
"abbr": "DOCE",
"venue": "",
"model": "LLM",
"pdf": "https://arxiv.org/pdf/2408.13745",
"code": "https://arxiv.org/pdf/2408.13745",
"year": 2024,
"group": "application"
},
{
"section": "Applications › Improve Generation Tasks › Code Generation",
"title": "USCD : Improving Code Generation of LLMs by Uncertainty-Aware Selective Contrastive Decoding",
"authors": "Shuai Wang, Liang Ding, Li Shen, Yong Luo, Zheng He, Wei Yu, Dacheng Tao",
"abbr": "USCD",
"venue": "",
"model": "LLM",
"pdf": "https://arxiv.org/pdf/2409.05923",
"code": "",
"year": 2024,
"group": "application"
},
{
"section": "Applications › Improve Generation Tasks › Code Generation",
"title": "Selective Prompt Anchoring for Code Generation",
"authors": "Yuan Tian, Tianyi Zhang",
"abbr": "SPA",
"venue": "",
"model": "LLM",
"pdf": "https://arxiv.org/pdf/2408.09121",
"code": "https://github.com/magic-YuanTian/Selective-Prompt-Anchoring",
"year": 2024,
"group": "application"
},
{
"section": "Applications › Improve Generation Tasks › Code Generation",
"title": "DocCGen: Document-based Controlled Code Generation",
"authors": "Sameer Pimparkhede, Mehant Kammakomati, Srikanth G. Tamilselvam, Prince Kumar, Ashok Pon Kumar, Pushpak Bhattacharyya",
"abbr": "DocCGen",
"venue": "EMNLP2024",
"model": "LLM",
"pdf": "https://aclanthology.org/2024.emnlp-main.1040.pdf",
"code": "https://github.com/sameerp30/Structured-generation",
"year": 2024,
"group": "application"
},
{
"section": "Applications › Improve Generation Tasks › Code Generation",
"title": "Constrained Decoding for Secure Code Generation",
"authors": "Yanjun Fu, Ethan Baker, Yu Ding, Yizheng Chen",
"abbr": "",
"venue": "",
"model": "LLM",
"pdf": "https://arxiv.org/pdf/2405.00218",
"code": "",
"year": 2024,
"group": "application"
},
{
"section": "Applications › Improve Generation Tasks › Code Generation",
"title": "Hot or Cold? Adaptive Temperature Sampling for Code Generation with Large Language Models",
"authors": "Yuqi Zhu, Jia Li, Ge Li, YunFei Zhao, Jia Li, Zhi Jin, Hong Mei",
"abbr": "AdapT",
"venue": "AAAI2024",
"model": "LLM",
"pdf": "https://arxiv.org/pdf/2309.02772",
"code": "https://github.com/LJ2lijia/AdapT",
"year": 2024,
"group": "application"
},
{
"section": "Applications › Improve Generation Tasks › Code Generation",
"title": "LEVER: Learning to Verify Language-to-Code Generation with Execution",
"authors": "Ansong Ni, Srini Iyer, Dragomir Radev, Ves Stoyanov, Wen-tau Yih, Sida I. Wang, Xi Victoria Lin",
"abbr": "LEVER",
"venue": "ICML2023",
"model": "LLM",
"pdf": "https://proceedings.mlr.press/v202/ni23b/ni23b.pdf",
"code": "https://github.com/niansong1996/lever",
"year": 2023,
"group": "application"
},
{
"section": "Applications › Improve Generation Tasks › Code Generation",
"title": "Decoding Secret Memorization in Code LLMs Through Token-Level Characterization",
"authors": "Yuqing Nie, Chong Wang, Kailong Wang, Guoai Xu, Guosheng Xu, Haoyu Wang",
"abbr": "DESEC",
"venue": "",
"model": "LLM",
"pdf": "https://arxiv.org/pdf/2410.08858",
"code": "",
"year": 2024,
"group": "application"
},
{
"section": "Applications › Improve Generation Efficiency › Text Generation",
"title": "Hierarchical Skip Decoding for Efficient Autoregressive Text Generation",
"authors": "Yunqi Zhu, Xuebing Yang, Yuanyuan Wu, Wensheng Zhang",
"abbr": "HSD",
"venue": "",
"model": "LLM",
"pdf": "https://arxiv.org/pdf/2403.14919",
"code": "",
"year": 2024,
"group": "application"
},
{
"section": "Applications › Improve Generation Efficiency › Text Generation",
"title": "A Frustratingly Simple Decoding Method for Neural Text Generation",
"authors": "Haoran Yang, Deng Cai, Huayang Li, Wei Bi, Wai Lam, Shuming Shi",
"abbr": "FSD",
"venue": "LREC2024",
"model": "LLM",
"pdf": "https://aclanthology.org/2024.lrec-main.47.pdf",
"code": "https://github.com/LHRYANG/FSD",
"year": 2024,
"group": "application"
},
{
"section": "Applications › Improve Generation Efficiency › Text Generation",
"title": "Adaptive Draft-Verification for Efficient Large Language Model Decoding",
"authors": "Xukun Liu, Bowen Lei, Ruqi Zhang, Dongkuan Xu",
"abbr": "ADED",
"venue": "",
"model": "LLM",
"pdf": "https://arxiv.org/pdf/2407.12021",
"code": "",
"year": 2024,
"group": "application"
},
{
"section": "Applications › Improve Generation Efficiency › Text Generation",
"title": "Position-Aware Depth Decay Decoding (D3): Boosting Large Language Model Inference Efficiency",
"authors": "Siqi Fan, Xuezhi Fang, Xingrun Xing, Peng Han, Shuo Shang, Yequan Wang",
"abbr": "D3",
"venue": "",
"model": "LLM",
"pdf": "https://arxiv.org/pdf/2503.08524",
"code": "",
"year": 2025,
"group": "application"
},
{
"section": "Applications › Improve Generation Efficiency › Image Generation",
"title": "Accelerating Auto-regressive Text-to-Image Generation with Training-free Speculative Jacobi Decoding",
"authors": "Yao Teng, Han Shi, Xian Liu, Xuefei Ning, Guohao Dai, Yu Wang, Zhenguo Li, Xihui Liu",
"abbr": "SJD",
"venue": "",
"model": "LVLM",
"pdf": "https://arxiv.org/pdf/2410.01699",
"code": "",
"year": 2024,
"group": "application"
},
{
"section": "Applications › Improve Generation Efficiency › Image Generation",
"title": "Emage: Non-Autoregressive Text-to-Image Generation",
"authors": "Zhangyin Feng, Runyi Hu, Liangxin Liu, Fan Zhang, Duyu Tang, Yong Dai, Xiaocheng Feng, Jiwei Li, Bing Qin, Shuming Shi",
"abbr": "Emage",
"venue": "",
"model": "LVLM",
"pdf": "https://arxiv.org/pdf/2312.14988",
"code": "",
"year": 2023,
"group": "application"
},
{
"section": "Applications › Improve Generation Efficiency › Image Generation",
"title": "HART: Efficient Visual Generation with Hybrid Autoregressive Transformer",
"authors": "Haotian Tang, Yecheng Wu, Shang Yang, Enze Xie, Junsong Chen, Junyu Chen, Zhuoyang Zhang, Han Cai, Yao Lu, Song Han",
"abbr": "HART",
"venue": "",
"model": "LVLM",
"pdf": "https://arxiv.org/pdf/2410.10812",
"code": "https://github.com/mit-han-lab/hart",
"year": 2024,
"group": "application"
},
{
"section": "Applications › Domain-Specific Applications › Healthcare",
"title": "Mitigating Hallucinations of Large Language Models in Medical Information Extraction via Contrastive Decoding",
"authors": "Derong Xu, Ziheng Zhang, Zhihong Zhu, Zhenxi Lin, Qidong Liu, Xian Wu, Tong Xu, Xiangyu Zhao, Yefeng Zheng, Enhong Chen",
"abbr": "ALCD",
"venue": "EMNLP2024-Findings",
"model": "LLM",
"pdf": "https://aclanthology.org/2024.findings-emnlp.456.pdf",
"code": "https://github.com/quqxui/quqxui-AlternateCD",
"year": 2024,
"group": "application"
},
{
"section": "Applications › Domain-Specific Applications › Robotics",
"title": "Grounded Decoding: Guiding Text Generation with Grounded Models for Embodied Agents",
"authors": "Wenlong Huang, Fei Xia, Dhruv Shah, Danny Driess, Andy Zeng, Yao Lu, Pete Florence, Igor Mordatch, Sergey Levine, Karol Hausman, Brian Ichter",
"abbr": "Grounded-Decoding",
"venue": "NeurIPS2023",
"model": "LLM",
"pdf": "https://arxiv.org/pdf/2303.00855",
"code": "https://grounded-decoding.github.io/",
"year": 2023,
"group": "application"
},
{
"section": "Applications › Domain-Specific Applications › Robotics",
"title": "Bidirectional Decoding: Improving Action Chunking via Closed-Loop Resampling",
"authors": "Yuejiang Liu, Jubayer Ibn Hamid, Annie Xie, Yoonho Lee, Maximilian Du, Chelsea Finn",
"abbr": "BidirectionalDecoding",
"venue": "",
"model": "LLM",
"pdf": "https://arxiv.org/pdf/2408.17355",
"code": "https://bid-robot.github.io/",
"year": 2024,
"group": "application"
}
]