Commit 6c268b4a by PolarisZZM

update integrated version

parent 6a4cb225
\begin{thebibliography}{238}
\providecommand{\natexlab}[1]{#1}
\providecommand{\url}[1]{\texttt{#1}}
\expandafter\ifx\csname urlstyle\endcsname\relax
\providecommand{\doi}[1]{doi: #1}\else
\providecommand{\doi}{doi: \begingroup \urlstyle{rm}\Url}\fi
\bibitem[ale()]{alexnet}
Krizhevsky A, Sutskever I, Hinton G E. ImageNet Classification with Deep
Convolutional Neural Networks. NeurIPS, 2012.
\bibitem[ali()]{align}
Jia C, et al. Scaling Up Visual and Vision-Language Representation Learning
With Noisy Text Supervision (ALIGN). ICML, 2021.
\bibitem[aud({\natexlab{a}})]{audiogen}
{\natexlab{a}}.
\newblock Kreuk F, et al. AudioGen: Textually Guided Audio Generation. ICLR,
2023.
\bibitem[aud({\natexlab{b}})]{audioldm}
{\natexlab{b}}.
\newblock Liu H, et al. AudioLDM: Text-to-Audio Generation with Latent
Diffusion Models. ICML, 2023.
\bibitem[aud({\natexlab{c}})]{audit}
{\natexlab{c}}.
\newblock Wang Y, et al. AUDIT: Audio Editing by Following Instructions with
Latent Diffusion Models. NeurIPS, 2023.
\bibitem[bei()]{beit}
Bao H, Dong L, Piao S, Wei F. BEiT: BERT Pre-Training of Image Transformers.
ICLR, 2022.
\bibitem[bli()]{blip2}
Li J, et al. BLIP-2: Bootstrapping Language-Image Pre-training with Frozen
Image Encoders and Large Language Models. ICML, 2023.
\bibitem[cfg()]{cfg}
Ho J, Salimans T. Classifier-Free Diffusion Guidance. arXiv:2207.12598, 2022.
\bibitem[cla()]{clap}
Elizalde B, Deshmukh S, Al Ismail M, Wang H. CLAP: Learning Audio Concepts From
Natural Language Supervision. ICASSP, 2023.
\bibitem[cli()]{clip}
Radford A, et al. Learning Transferable Visual Models From Natural Language
Supervision (CLIP). ICML, 2021.
\bibitem[cod()]{codi}
Tang Z, et al. Any-to-Any Generation via Composable Diffusion (CoDi). NeurIPS,
2023.
\bibitem[con()]{controlnet}
Zhang L, Rao A, Agrawala M. Adding Conditional Control to Text-to-Image
Diffusion Models (ControlNet). ICCV, 2023.
\bibitem[dal({\natexlab{a}})]{dalle}
{\natexlab{a}}.
\newblock Ramesh A, et al. Zero-Shot Text-to-Image Generation (DALL·E). ICML,
2021.
\bibitem[dal({\natexlab{b}})]{dalle2}
{\natexlab{b}}.
\newblock Ramesh A, et al. Hierarchical Text-Conditional Image Generation with
CLIP Latents (DALL·E 2). arXiv:2204.06125, 2022.
\bibitem[ddp()]{ddpm}
Ho J, Jain A, Abbeel P. Denoising Diffusion Probabilistic Models (DDPM).
NeurIPS, 2020.
\bibitem[din()]{dino}
Caron M, et al. Emerging Properties in Self-Supervised Vision Transformers
(DINO). ICCV, 2021.
\bibitem[dre()]{dreambooth}
Ruiz N, et al. DreamBooth: Fine Tuning Text-to-Image Diffusion Models for
Subject-Driven Generation. CVPR, 2023.
\bibitem[emu()]{emu}
Sun Q, et al. Emu: Generative Pretraining in Multimodality. ICLR, 2024.
\bibitem[fas()]{fastspeech2}
Ren Y, et al. FastSpeech 2: Fast and High-Quality End-to-End Text to Speech.
ICLR, 2021.
\bibitem[fla()]{flamingo}
Alayrac J B, et al. Flamingo: a Visual Language Model for Few-Shot Learning.
NeurIPS, 2022.
\bibitem[gan()]{gan}
Goodfellow I, et al. Generative Adversarial Nets. NeurIPS, 2014.
\bibitem[gli()]{glide}
Nichol A, et al. GLIDE: Towards Photorealistic Image Generation and Editing
with Text-Guided Diffusion Models. ICML, 2022.
\bibitem[hub()]{hubert}
Hsu W N, et al. HuBERT: Self-Supervised Speech Representation Learning by
Masked Prediction of Hidden Units. IEEE/ACM Transactions on Audio, Speech,
and Language Processing, 2021.
\bibitem[ima()]{imagen}
Saharia C, et al. Photorealistic Text-to-Image Diffusion Models with Deep
Language Understanding (Imagen). NeurIPS, 2022.
\bibitem[ins()]{instructblip}
Dai W, et al. InstructBLIP: Towards General-purpose Vision-Language Models with
Instruction Tuning. NeurIPS, 2023.
\bibitem[ip2()]{ip2p}
Brooks T, Holynski A, Efros A A. InstructPix2Pix: Learning to Follow Image
Editing Instructions. CVPR, 2023.
\bibitem[ldm()]{ldm}
Rombach R, et al. High-Resolution Image Synthesis with Latent Diffusion Models
(Stable Diffusion). CVPR, 2022.
\bibitem[lla()]{llava}
Liu H, et al. Visual Instruction Tuning (LLaVA). NeurIPS, 2023.
\bibitem[mae()]{mae}
He K, et al. Masked Autoencoders Are Scalable Vision Learners (MAE). CVPR,
2022.
\bibitem[mak({\natexlab{a}})]{makeanaudio}
{\natexlab{a}}.
\newblock Huang R, et al. Make-An-Audio: Text-To-Audio Generation with
Prompt-Enhanced Diffusion Models. ICML, 2023.
\bibitem[mak({\natexlab{b}})]{makeanaudio2}
{\natexlab{b}}.
\newblock Huang J, et al. Make-An-Audio 2: Temporal-Enhanced Text-to-Audio
Generation. arXiv:2305.18474, 2023.
\bibitem[moc()]{moco}
He K, et al. Momentum Contrast for Unsupervised Visual Representation Learning
(MoCo). CVPR, 2020.
\bibitem[nis()]{nist-synthetic}
National Institute of Standards and Technology. Reducing Risks Posed by
Synthetic Content. NIST AI 100-4, 2024.
\bibitem[p2p()]{p2p}
Hertz A, et al. Prompt-to-Prompt Image Editing with Cross Attention Control.
ICLR, 2023.
\bibitem[qwe({\natexlab{a}})]{qwenaudio}
{\natexlab{a}}.
\newblock Chu Y, et al. Qwen-Audio: Advancing Universal Audio Understanding via
Unified Large-Scale Audio-Language Models. arXiv:2311.07919, 2023.
\bibitem[qwe({\natexlab{b}})]{qwenvl}
{\natexlab{b}}.
\newblock Bai J, et al. Qwen-VL: A Versatile Vision-Language Model for
Understanding, Localization, Text Reading, and Beyond. arXiv:2308.12966,
2023.
\bibitem[res()]{resnet}
He K, Zhang X, Ren S, Sun J. Deep Residual Learning for Image Recognition.
CVPR, 2016.
\bibitem[sal()]{salmonn}
Tang C, et al. SALMONN: Towards Generic Hearing Abilities for Large Language
Models. ICLR, 2024.
\bibitem[sde()]{sdedit}
Meng C, et al. SDEdit: Guided Image Synthesis and Editing with Stochastic
Differential Equations. ICLR, 2022.
\bibitem[sig()]{siglip}
Zhai X, et al. Sigmoid Loss for Language Image Pre-Training (SigLIP). ICCV,
2023.
\bibitem[sim()]{simclr}
Chen T, et al. A Simple Framework for Contrastive Learning of Visual
Representations (SimCLR). ICML, 2020.
\bibitem[spe()]{speechgpt}
Zhang D, et al. SpeechGPT: Empowering Large Language Models with Intrinsic
Cross-Modal Conversational Abilities. Findings of EMNLP, 2023.
\bibitem[tac()]{tacotron2}
Shen J, et al. Natural TTS Synthesis by Conditioning WaveNet on Mel Spectrogram
Predictions (Tacotron 2). ICASSP, 2018.
\bibitem[vae()]{vae}
Kingma D P, Welling M. Auto-Encoding Variational Bayes (VAE). ICLR, 2014.
\bibitem[val()]{valle}
Wang C, et al. Neural Codec Language Models are Zero-Shot Text to Speech
Synthesizers (VALL-E). arXiv:2301.02111, 2023.
\bibitem[vgg()]{vgg}
Simonyan K, Zisserman A. Very Deep Convolutional Networks for Large-Scale Image
Recognition (VGG). ICLR, 2015.
\bibitem[vit({\natexlab{a}})]{vit}
{\natexlab{a}}.
\newblock Dosovitskiy A, et al. An Image is Worth 16x16 Words: Transformers for
Image Recognition at Scale. ICLR, 2021.
\bibitem[vit({\natexlab{b}})]{vits}
{\natexlab{b}}.
\newblock Kim J, Kong J, Son J. Conditional Variational Autoencoder with
Adversarial Learning for End-to-End Text-to-Speech (VITS). ICML, 2021.
\bibitem[wac()]{waco}
Ouyang S, Ye R, Li L. WACO: Word-Aligned Contrastive Learning for Speech
Translation. ACL, 2023.
\bibitem[wav()]{wav2vec2}
Baevski A, Zhou H, Mohamed A, Auli M. wav2vec 2.0: A Framework for
Self-Supervised Learning of Speech Representations. NeurIPS, 2020.
\bibitem[whi()]{whisper}
Radford A, et al. Robust Speech Recognition via Large-Scale Weak Supervision
(Whisper). ICML, 2023.
\bibitem[yao(2023)]{yao-etal:2023tree}
Tree of thoughts, 2023.
\newblock Placeholder entry added during textbook merge; replace with full
bibliographic metadata.
\bibitem[zhe(2023)]{zheng-etal:2023judging}
Judging large language model reasoning with generative verification, 2023.
\newblock Placeholder entry added during textbook merge; replace with full
bibliographic metadata.
\bibitem[wan(2024)]{wang-etal:2024math}
Math reasoning verification with rollout, 2024.
\newblock Placeholder entry added during textbook merge; replace with full
bibliographic metadata.
\bibitem[Aky{\"u}rek et~al.(2023)Aky{\"u}rek, Schuurmans, Andreas, Ma, and
Zhou]{akyurek-etal:2023what}
Ekin Aky{\"u}rek, Dale Schuurmans, Jacob Andreas, Tengyu Ma, and Denny Zhou.
\newblock What learning algorithm is in-context learning? investigations with
linear models.
\newblock In \emph{Proceedings of The Eleventh International Conference on
Learning Representations}, 2023.
\bibitem[Allal et~al.(2024)Allal, Lozhkov, and van
Strien]{allal-etal:2024cosmopedia}
Loubna~Ben Allal, Anton Lozhkov, and Daniel van Strien.
\newblock cosmopedia: how to create large-scale synthetic data for
pre-training.
\newblock \url{https://huggingface.co/blog/cosmopedia}, 2024.
\bibitem[Anderson(1982)]{anderson:1982reverse}
Brian~D.O. Anderson.
\newblock Reverse-time diffusion equation models.
\newblock \emph{Stochastic Processes and their Applications}, 12\penalty0
(3):\penalty0 313--326, 1982.
\bibitem[Austin et~al.(2021)Austin, Johnson, Ho, Tarlow, and Van
Den~Berg]{austin-etal:2021structured}
Jacob Austin, Daniel~D Johnson, Jonathan Ho, Daniel Tarlow, and Rianne Van
Den~Berg.
\newblock Structured denoising diffusion models in discrete state-spaces.
\newblock \emph{Advances in neural information processing systems},
34:\penalty0 17981--17993, 2021.
\bibitem[Ba et~al.(2016)Ba, Kiros, and Hinton]{BaEtAl2016LayerNorm}
Lei~Jimmy Ba, Jamie~Ryan Kiros, and Geoffrey~E. Hinton.
\newblock Layer normalization.
\newblock \emph{arXiv preprint arXiv:1607.06450}, 2016.
\newblock URL \url{https://arxiv.org/abs/1607.06450}.
\bibitem[Bach et~al.(2022)Bach, Sanh, Yong, Webson, Raffel, Nayak, Sharma, Kim,
Bari, F{\'{e}}vry, Alyafeai, Dey, Santilli, Sun, Ben{-}David, Xu, Chhablani,
Wang, Fries, AlShaibani, Sharma, Thakker, Almubarak, Tang, Radev, Jiang, and
Rush]{bach-etal:2022promptsource}
Stephen~H. Bach, Victor Sanh, Zheng~Xin Yong, Albert Webson, Colin Raffel,
Nihal~V. Nayak, Abheesht Sharma, Taewoon Kim, M.~Saiful Bari, Thibault
F{\'{e}}vry, Zaid Alyafeai, Manan Dey, Andrea Santilli, Zhiqing Sun, Srulik
Ben{-}David, Canwen Xu, Gunjan Chhablani, Han Wang, Jason~Alan Fries,
Maged~Saeed AlShaibani, Shanya Sharma, Urmish Thakker, Khalid Almubarak,
Xiangru Tang, Dragomir~R. Radev, Mike~Tian{-}Jian Jiang, and Alexander~M.
Rush.
\newblock Promptsource: An integrated development environment and repository
for natural language prompts.
\newblock In \emph{Proceedings of the 60th Annual Meeting of the Association
for Computational Linguistics: System Demonstrations}, pages 93--104, 2022.
\bibitem[Baevski et~al.(2020)Baevski, Zhou, Mohamed, and
Auli]{BaevskiEtAl2020Wav2vec}
Alexei Baevski, Yuhao Zhou, Abdelrahman Mohamed, and Michael Auli.
\newblock wav2vec 2.0: A framework for self-supervised learning of speech
representations.
\newblock In \emph{Advances in Neural Information Processing Systems 33}, 2020.
\newblock URL
\url{https://proceedings.neurips.cc/paper/2020/hash/92d1e1eb1cd6f9fba3227870bb6d7f07-Abstract.html}.
\bibitem[Bahdanau et~al.(2015)Bahdanau, Cho, and
Bengio]{BahdanauEtAl2015Attention}
Dzmitry Bahdanau, Kyunghyun Cho, and Yoshua Bengio.
\newblock Neural machine translation by jointly learning to align and
translate.
\newblock In \emph{Proceedings of the 3rd International Conference on Learning
Representations}, 2015.
\newblock URL \url{https://arxiv.org/abs/1409.0473}.
\bibitem[Bengio et~al.(2000)Bengio, Ducharme, and
Vincent]{bengio-etal:2000neural}
Yoshua Bengio, R{\'e}jean Ducharme, and Pascal Vincent.
\newblock A neural probabilistic language model.
\newblock \emph{Advances in Neural Information Processing Systems}, 13, 2000.
\bibitem[Bengio et~al.(2003)Bengio, Ducharme, Vincent, and
Janvin]{BengioEtAl2003NeuralLM}
Yoshua Bengio, R{\'e}jean Ducharme, Pascal Vincent, and Christian Janvin.
\newblock A neural probabilistic language model.
\newblock \emph{Journal of Machine Learning Research}, 3:\penalty0 1137--1155,
2003.
\newblock URL \url{https://jmlr.org/papers/v3/bengio03a.html}.
\bibitem[Bengio et~al.(2006)Bengio, Lamblin, Popovici, and
Larochelle]{bengio-etal:2006greedy}
Yoshua Bengio, Pascal Lamblin, Dan Popovici, and Hugo Larochelle.
\newblock Greedy layer-wise training of deep networks.
\newblock \emph{Advances in Neural Information Processing Systems}, 19, 2006.
\bibitem[Bengio et~al.(2013)Bengio, Courville, and
Vincent]{BengioEtAl2013Representation}
Yoshua Bengio, Aaron~C. Courville, and Pascal Vincent.
\newblock Representation learning: A review and new perspectives.
\newblock \emph{IEEE Transactions on Pattern Analysis and Machine
Intelligence}, 35\penalty0 (8):\penalty0 1798--1828, 2013.
\newblock \doi{10.1109/TPAMI.2013.50}.
\bibitem[Bishop(1995)]{bishop:1995training}
Christopher~M. Bishop.
\newblock Training with noise is equivalent to {Tikhonov} regularization.
\newblock \emph{Neural Computation}, 7\penalty0 (1):\penalty0 108--116, 1995.
\bibitem[Boyd and Vandenberghe(2004)]{BoydVandenberghe2004Convex}
Stephen Boyd and Lieven Vandenberghe.
\newblock \emph{Convex Optimization}.
\newblock Cambridge University Press, 2004.
\newblock ISBN 978-0-521-83378-3.
\newblock URL \url{https://web.stanford.edu/~boyd/cvxbook/}.
\bibitem[Bradley and Terry(1952)]{bradley-and-terry:rank}
Ralph~Allan Bradley and Milton~E. Terry.
\newblock Rank analysis of incomplete block designs: I. the method of paired
comparisons.
\newblock \emph{Biometrika}, 39\penalty0 (3/4):\penalty0 324--345, 1952.
\bibitem[Brown et~al.(1993)Brown, Della~Pietra, Della~Pietra, and
Mercer]{brown-etal:1993mathematics}
Peter~F. Brown, Stephen~A. Della~Pietra, Vincent~J. Della~Pietra, and Robert~L.
Mercer.
\newblock The mathematics of statistical machine translation: Parameter
estimation.
\newblock \emph{Computational Linguistics}, 19\penalty0 (2):\penalty0 263--311,
1993.
\bibitem[Brown et~al.(2020{\natexlab{a}})Brown, Mann, Ryder, Subbiah, Kaplan,
Dhariwal, Neelakantan, Shyam, Sastry, Askell, Agarwal, Herbert-Voss, Krueger,
Henighan, Child, Ramesh, Ziegler, Wu, Winter, Hesse, Chen, Sigler, Litwin,
Gray, Chess, Clark, Berner, McCandlish, Radford, Sutskever, and
Amodei]{brown-etal:2020language}
Tom Brown, Benjamin Mann, Nick Ryder, Melanie Subbiah, Jared~D Kaplan, Prafulla
Dhariwal, Arvind Neelakantan, Pranav Shyam, Girish Sastry, Amanda Askell,
Sandhini Agarwal, Ariel Herbert-Voss, Gretchen Krueger, Tom Henighan, Rewon
Child, Aditya Ramesh, Daniel Ziegler, Jeffrey Wu, Clemens Winter, Chris
Hesse, Mark Chen, Eric Sigler, Mateusz Litwin, Scott Gray, Benjamin Chess,
Jack Clark, Christopher Berner, Sam McCandlish, Alec Radford, Ilya Sutskever,
and Dario Amodei.
\newblock Language models are few-shot learners.
\newblock \emph{Advances in neural information processing systems},
33:\penalty0 1877--1901, 2020{\natexlab{a}}.
\bibitem[Brown et~al.(2020{\natexlab{b}})Brown, Mann, Ryder, Subbiah, Kaplan,
Dhariwal, Neelakantan, Shyam, Sastry, Askell, Agarwal, Herbert{-}Voss,
Krueger, Henighan, Child, Ramesh, Ziegler, Wu, Winter, Hesse, Chen, Sigler,
Litwin, Gray, Chess, Clark, Berner, McCandlish, Radford, Sutskever, and
Amodei]{BrownEtAl2020GPT3}
Tom~B. Brown, Benjamin Mann, Nick Ryder, Melanie Subbiah, Jared Kaplan,
Prafulla Dhariwal, Arvind Neelakantan, Pranav Shyam, Girish Sastry, Amanda
Askell, Sandhini Agarwal, Ariel Herbert{-}Voss, Gretchen Krueger, Tom
Henighan, Rewon Child, Aditya Ramesh, Daniel~M. Ziegler, Jeffrey Wu, Clemens
Winter, Christopher Hesse, Mark Chen, Eric Sigler, Mateusz Litwin, Scott
Gray, Benjamin Chess, Jack Clark, Christopher Berner, Sam McCandlish, Alec
Radford, Ilya Sutskever, and Dario Amodei.
\newblock Language models are few-shot learners.
\newblock In \emph{Advances in Neural Information Processing Systems 33}, pages
1877--1901, 2020{\natexlab{b}}.
\newblock URL
\url{https://proceedings.neurips.cc/paper/2020/hash/1457c0d6bfcb4967418bfb8ac142f64a-Abstract.html}.
\bibitem[Bubeck et~al.(2023)Bubeck, Chandrasekaran, Eldan, Gehrke, Horvitz,
Kamar, Lee, Lee, Li, Lundberg, Nori, Palangi, Ribeiro, and
Zhang]{bubeck-etal:2023sparks}
S{\'{e}}bastien Bubeck, Varun Chandrasekaran, Ronen Eldan, Johannes Gehrke,
Eric Horvitz, Ece Kamar, Peter Lee, Yin~Tat Lee, Yuanzhi Li, Scott~M.
Lundberg, Harsha Nori, Hamid Palangi, Marco~T{\'{u}}lio Ribeiro, and
Yi~Zhang.
\newblock Sparks of artificial general intelligence: Early experiments with
gpt-4.
\newblock \emph{arXiv preprint arXiv:2303.12712}, 2023.
\bibitem[Campbell et~al.(2024)Campbell, Yim, Barzilay, Rainforth, and
Jaakkola]{campbell-etal:2024generative}
Andrew Campbell, Jason Yim, Regina Barzilay, Tom Rainforth, and Tommi Jaakkola.
\newblock Generative flows on discrete state-spaces: Enabling multimodal flows
with applications to protein co-design.
\newblock In \emph{International Conference on Machine Learning}, pages
5453--5512. PMLR, 2024.
\bibitem[Charniak(1997)]{charniak:1997statistical}
Eugene Charniak.
\newblock Statistical parsing with a context-free grammar and word statistics.
\newblock \emph{AAAI/IAAI}, 2005\penalty0 (598-603):\penalty0 18, 1997.
\bibitem[Chaudhari et~al.(2021)Chaudhari, Mithal, Polatkan, and
Ramanath]{chaudhari-etal:2021attentive}
Sneha Chaudhari, Varun Mithal, Gungor Polatkan, and Rohan Ramanath.
\newblock An attentive survey of attention models.
\newblock \emph{ACM Transactions on Intelligent Systems and Technology},
12\penalty0 (5):\penalty0 1--32, 2021.
\bibitem[Chen et~al.(2024{\natexlab{a}})Chen, Li, Yan, Wang, Gunaratna, Yadav,
Tang, Srinivasan, Zhou, Huang, and Jin]{chen-etal:2024alpagasus}
Lichang Chen, Shiyang Li, Jun Yan, Hai Wang, Kalpa Gunaratna, Vikas Yadav,
Zheng Tang, Vijay Srinivasan, Tianyi Zhou, Heng Huang, and Hongxia Jin.
\newblock Alpagasus: Training a better alpaca with fewer data.
\newblock In \emph{The Twelfth International Conference on Learning
Representations}, 2024{\natexlab{a}}.
\bibitem[Chen and Goodman(1999)]{Chen-and-Goodman:1999}
Stanley~F. Chen and Joshua Goodman.
\newblock An empirical study of smoothing techniques for language modeling.
\newblock \emph{Computer Speech and Language}, 13:\penalty0 359--394, 1999.
\bibitem[Chen et~al.(2020)Chen, Kornblith, Norouzi, and
Hinton]{ChenEtAl2020SimCLR}
Ting Chen, Simon Kornblith, Mohammad Norouzi, and Geoffrey~E. Hinton.
\newblock A simple framework for contrastive learning of visual
representations.
\newblock In \emph{Proceedings of the 37th International Conference on Machine
Learning}, volume 119 of \emph{Proceedings of Machine Learning Research},
pages 1597--1607. PMLR, 2020.
\newblock URL \url{https://proceedings.mlr.press/v119/chen20j.html}.
\bibitem[Chen et~al.(2024{\natexlab{b}})Chen, Deng, Yuan, Ji, and
Gu]{chen-etal:2024self}
Zixiang Chen, Yihe Deng, Huizhuo Yuan, Kaixuan Ji, and Quanquan Gu.
\newblock Self-play fine-tuning converts weak language models to strong
language models.
\newblock \emph{arXiv preprint arXiv:2401.01335}, 2024{\natexlab{b}}.
\bibitem[Chevalier et~al.(2023)Chevalier, Wettig, Ajith, and
Chen]{chevalier-etal:2023adapting}
Alexis Chevalier, Alexander Wettig, Anirudh Ajith, and Danqi Chen.
\newblock Adapting language models to compress contexts.
\newblock In \emph{Proceedings of the 2023 Conference on Empirical Methods in
Natural Language Processing}, pages 3829--3846, 2023.
\bibitem[Chiang et~al.(2023)Chiang, Li, Lin, Sheng, Wu, Zhang, Zheng, Zhuang,
Zhuang, Gonzalez, Stoica, and Xing]{chiang-etal:2023vicuna}
Wei-Lin Chiang, Zhuohan Li, Zi~Lin, Ying Sheng, Zhanghao Wu, Hao Zhang, Lianmin
Zheng, Siyuan Zhuang, Yonghao Zhuang, Joseph~E. Gonzalez, Ion Stoica, and
Eric~P. Xing.
\newblock Vicuna: An open-source chatbot impressing gpt-4 with 90\%* chatgpt
quality, March 2023.
\newblock URL \url{https://lmsys.org/blog/2023-03-30-vicuna/}.
\bibitem[Cho et~al.(2014)Cho, van Merri{\"e}nboer, Gulcehre, Bahdanau,
Bougares, Schwenk, and Bengio]{cho-etal:2014learning}
Kyunghyun Cho, Bart van Merri{\"e}nboer, Caglar Gulcehre, Dzmitry Bahdanau,
Fethi Bougares, Holger Schwenk, and Yoshua Bengio.
\newblock Learning phrase representations using {RNN} encoder--decoder for
statistical machine translation.
\newblock In \emph{Proceedings of the 2014 Conference on Empirical Methods in
Natural Language Processing}, pages 1724--1734, 2014.
\bibitem[Chu et~al.(2023)Chu, Chen, Chen, Yu, He, Wang, Peng, Liu, Qin, and
Liu]{chu-etal:2023survey}
Zheng Chu, Jingchang Chen, Qianglong Chen, Weijiang Yu, Tao He, Haotian Wang,
Weihua Peng, Ming Liu, Bing Qin, and Ting Liu.
\newblock A survey of chain of thought reasoning: Advances, frontiers and
future.
\newblock \emph{arXiv preprint arXiv:2309.15402}, 2023.
\bibitem[Chung et~al.(2022)Chung, Hou, Longpre, Zoph, Tay, Fedus, Li, Wang,
Dehghani, Brahma, Webson, Gu, Dai, Suzgun, Chen, Chowdhery, Valter, Narang,
Mishra, Yu, Zhao, Huang, Dai, Yu, Petrov, Chi, Dean, Devlin, Roberts, Zhou,
Le, and Wei]{chung-etal:2022scaling}
Hyung~Won Chung, Le~Hou, S.~Longpre, Barret Zoph, Yi~Tay, William Fedus, Eric
Li, Xuezhi Wang, Mostafa Dehghani, Siddhartha Brahma, Albert Webson,
Shixiang~Shane Gu, Zhuyun Dai, Mirac Suzgun, Xinyun Chen, Aakanksha
Chowdhery, Dasha Valter, Sharan Narang, Gaurav Mishra, Adams~Wei Yu, Vincent
Zhao, Yanping Huang, Andrew~M. Dai, Hongkun Yu, Slav Petrov, Ed~Huai~hsin
Chi, Jeff Dean, Jacob Devlin, Adam Roberts, Denny Zhou, Quoc~V. Le, and Jason
Wei.
\newblock Scaling instruction-finetuned language models.
\newblock \emph{arXiv preprint arXiv:2210.11416}, 2022.
\bibitem[Clark et~al.(2019)Clark, Luong, Le, and
Manning]{clark-etal:2019electra}
Kevin Clark, Minh-Thang Luong, Quoc~V Le, and Christopher~D Manning.
\newblock Electra: Pre-training text encoders as discriminators rather than
generators.
\newblock In \emph{Proceedings of International Conference on Learning
Representations}, 2019.
\bibitem[Collobert and Weston(2008)]{Collobert-and-Weston:2008AUA}
Ronan Collobert and Jason Weston.
\newblock A unified architecture for natural language processing: Deep neural
networks with multitask learning.
\newblock In \emph{Proceedings of the 25th International Conference on Machine
Learning}, pages 160--167, 2008.
\bibitem[Cortes and Vapnik(1995)]{Cortes-and-Vapnik:1995}
Corinna Cortes and Vladimir Vapnik.
\newblock Support-vector networks.
\newblock \emph{Machine Learning}, 20:\penalty0 273--297, 1995.
\bibitem[Cui et~al.(2024)Cui, Yuan, Ding, Yao, He, Zhu, Ni, Xie, Xie, Lin, Liu,
and Sun]{cui-etal:2024ultra}
Ganqu Cui, Lifan Yuan, Ning Ding, Guanming Yao, Bingxiang He, Wei Zhu, Yuan Ni,
Guotong Xie, Ruobing Xie, Yankai Lin, Zhiyuan Liu, and Maosong Sun.
\newblock {ULTRAFEEDBACK}: Boosting language models with scaled {AI} feedback.
\newblock In \emph{Proceedings of the 41st International Conference on Machine
Learning}, volume 235, pages 9722--9744, 2024.
\bibitem[Dai et~al.(2023)Dai, Sun, Dong, Hao, Ma, Sui, and
Wei]{dai-etal:2023can}
Damai Dai, Yutao Sun, Li~Dong, Yaru Hao, Shuming Ma, Zhifang Sui, and Furu Wei.
\newblock Why can gpt learn in-context? language models secretly perform
gradient descent as meta-optimizers.
\newblock In \emph{Findings of the Association for Computational Linguistics:
ACL 2023}, pages 4005--4019, 2023.
\bibitem[Dettmers et~al.(2023)Dettmers, Pagnoni, Holtzman, and
Zettlemoyer]{dettmers-etal:2023qlora}
Tim Dettmers, Artidoro Pagnoni, Ari Holtzman, and Luke Zettlemoyer.
\newblock Qlora: Efficient finetuning of quantized llms.
\newblock In \emph{Advances in Neural Information Processing Systems
(NeurIPS)}, 2023.
\bibitem[Devlin et~al.(2019{\natexlab{a}})Devlin, Chang, Lee, and
Toutanova]{DevlinEtAl2019BERT}
Jacob Devlin, Ming{-}Wei Chang, Kenton Lee, and Kristina Toutanova.
\newblock {BERT}: Pre-training of deep bidirectional transformers for language
understanding.
\newblock In \emph{Proceedings of the 2019 Conference of the North American
Chapter of the Association for Computational Linguistics: Human Language
Technologies}, pages 4171--4186. Association for Computational Linguistics,
2019{\natexlab{a}}.
\newblock \doi{10.18653/v1/N19-1423}.
\bibitem[Devlin et~al.(2019{\natexlab{b}})Devlin, Chang, Lee, and
Toutanova]{devlin-etal:2019bert}
Jacob Devlin, Ming-Wei Chang, Kenton Lee, and Kristina Toutanova.
\newblock Bert: Pre-training of deep bidirectional transformers for language
understanding.
\newblock In \emph{Proceedings of the 2019 Conference of the North American
Chapter of the Association for Computational Linguistics: Human Language
Technologies, Volume 1 (Long and Short Papers)}, pages 4171--4186,
2019{\natexlab{b}}.
\bibitem[Dror et~al.(2018)Dror, Baumer, Shlomov, and
Reichart]{DrorEtAl2018Significance}
Rotem Dror, Gili Baumer, Segev Shlomov, and Roi Reichart.
\newblock The hitchhiker's guide to testing statistical significance in natural
language processing.
\newblock In \emph{Proceedings of the 56th Annual Meeting of the Association
for Computational Linguistics}, pages 1383--1392. Association for
Computational Linguistics, 2018.
\newblock \doi{10.18653/v1/P18-1128}.
\bibitem[Dror et~al.(2020)Dror, Peled-Cohen, and Shlomov]{Dror-et-al:2020}
Rotem Dror, Lotem Peled-Cohen, and Segev Shlomov.
\newblock \emph{Neural Network Methods for Natural Language Processing}.
\newblock Morgan \& Claypool Publishers, 2020.
\bibitem[Dubey et~al.(2024)Dubey, Jauhri, Pandey, Kadian, Al-Dahle, Letman,
Mathur, Schelten, Yang, Fan, et~al.]{dubey2024llama}
Abhimanyu Dubey, Abhinav Jauhri, Abhinav Pandey, Abhishek Kadian, Ahmad
Al-Dahle, Aiesha Letman, Akhil Mathur, Alan Schelten, Amy Yang, Angela Fan,
et~al.
\newblock The llama 3 herd of models.
\newblock \emph{arXiv preprint arXiv:2407.21783}, 2024.
\bibitem[Dubois et~al.(2024)Dubois, Li, Taori, Zhang, Gulrajani, Ba, Guestrin,
Liang, and Hashimoto]{dubois-etal:2024alpacafarm}
Yann Dubois, Chen~Xuechen Li, Rohan Taori, Tianyi Zhang, Ishaan Gulrajani,
Jimmy Ba, Carlos Guestrin, Percy~S Liang, and Tatsunori~B Hashimoto.
\newblock Alpacafarm: A simulation framework for methods that learn from human
feedback.
\newblock \emph{Advances in Neural Information Processing Systems}, 36, 2024.
\bibitem[Duchi et~al.(2011)Duchi, Hazan, and Singer]{duchi-etal:2011adaptive}
John Duchi, Elad Hazan, and Yoram Singer.
\newblock Adaptive subgradient methods for online learning and stochastic
optimization.
\newblock \emph{Journal of Machine Learning Research}, 12:\penalty0 2121--2159,
2011.
\bibitem[Erhan et~al.(2010)Erhan, Courville, Bengio, and
Vincent]{erhan-etal:2010does}
Dumitru Erhan, Aaron Courville, Yoshua Bengio, and Pascal Vincent.
\newblock Why does unsupervised pre-training help deep learning?
\newblock In \emph{Proceedings of the Thirteenth International Conference on
Artificial Intelligence and Statistics}, pages 201--208, 2010.
\bibitem[Esser et~al.(2024)Esser, Kulal, Blattmann, Entezari, M{\" u}ller,
Saini, Levi, Lorenz, Sauer, Boesel, Podell, Dockhorn, English, Lacey,
Goodwin, Marek, and Rombach]{esser-etal:2024scaling}
Patrick Esser, Sumith Kulal, Andreas Blattmann, Rahim Entezari, Jonas M{\"
u}ller, Harry Saini, Yam Levi, Dominik Lorenz, Axel Sauer, Frederic Boesel,
Dustin Podell, Tim Dockhorn, Zion English, Kyle Lacey, Alex Goodwin, Yannik
Marek, and Robin Rombach.
\newblock Scaling rectified flow transformers for high-resolution image
synthesis.
\newblock In \emph{Forty-first international conference on machine learning},
2024.
\bibitem[Feng et~al.(2021)Feng, Gangal, Wei, Chandar, Vosoughi, Mitamura, and
Hovy]{feng-etal:2021survey}
Steven~Y. Feng, Varun Gangal, Jason Wei, Sarath Chandar, Soroush Vosoughi,
Teruko Mitamura, and Eduard Hovy.
\newblock A survey of data augmentation approaches for {NLP}.
\newblock In \emph{Findings of the Association for Computational Linguistics:
ACL-IJCNLP 2021}, pages 968--988, 2021.
\bibitem[Firth(1957)]{firth:1957synopsis}
John~R. Firth.
\newblock A synopsis of linguistic theory, 1930--1955.
\newblock \emph{Studies in Linguistic Analysis}, 1957.
\bibitem[Freedman et~al.(2007)Freedman, Pisani, and
Purves]{Freedman-et-al-2007}
David Freedman, Robert Pisani, and Roger Purves.
\newblock \emph{Statistics}.
\newblock W. W. Norton \& Company, 4 edition, 2007.
\bibitem[Freedman(2009)]{Freedman:2009}
David~A. Freedman.
\newblock \emph{Statistical Models: Theory and Practice}.
\newblock Cambridge University Press, 2 edition, 2009.
\bibitem[Garg et~al.(2022)Garg, Tsipras, Liang, and Valiant]{garg-etal:2022can}
Shivam Garg, Dimitris Tsipras, Percy~S Liang, and Gregory Valiant.
\newblock What can transformers learn in-context? a case study of simple
function classes.
\newblock \emph{Advances in Neural Information Processing Systems},
35:\penalty0 30583--30598, 2022.
\bibitem[Gat et~al.(2024)Gat, Remez, Shaul, Kreuk, Chen, Synnaeve, Adi, and
Lipman]{gat-etal:2024discrete}
Itai Gat, Tal Remez, Neta Shaul, Felix Kreuk, Ricky~TQ Chen, Gabriel Synnaeve,
Yossi Adi, and Yaron Lipman.
\newblock Discrete flow matching.
\newblock \emph{Advances in Neural Information Processing Systems},
37:\penalty0 133345--133385, 2024.
\bibitem[Ghazvininejad et~al.(2019)Ghazvininejad, Levy, Liu, and
Zettlemoyer]{ghazvininejad-etal:2019mask}
Marjan Ghazvininejad, Omer Levy, Yinhan Liu, and Luke Zettlemoyer.
\newblock Mask-predict: Parallel decoding of conditional masked language
models.
\newblock In \emph{Proceedings of the 2019 Conference on Empirical Methods in
Natural Language Processing and the 9th International Joint Conference on
Natural Language Processing (EMNLP-IJCNLP)}, pages 6112--6121, 2019.
\bibitem[Gillespie(1977)]{gillespie:1977exact}
Daniel~T Gillespie.
\newblock Exact stochastic simulation of coupled chemical reactions.
\newblock \emph{The journal of physical chemistry}, 81\penalty0 (25):\penalty0
2340--2361, 1977.
\bibitem[Gillespie(2001)]{gillespie:2001approximate}
Daniel~T Gillespie.
\newblock Approximate accelerated stochastic simulation of chemically reacting
systems.
\newblock \emph{The Journal of chemical physics}, 115\penalty0 (4):\penalty0
1716--1733, 2001.
\bibitem[Gong et~al.(2023)Gong, Li, Feng, Wu, and Kong]{gong-etal:2023diffuseq}
Shansan Gong, Mukai Li, Jiangtao Feng, Zhiyong Wu, and Lingpeng Kong.
\newblock Diffuseq: Sequence to sequence text generation with diffusion models.
\newblock In \emph{The Eleventh International Conference on Learning
Representations}, 2023.
\bibitem[Goodfellow et~al.(2015)Goodfellow, Shlens, and
Szegedy]{Goodfellow-etal:2015Adversarial}
Ian Goodfellow, Jonathon Shlens, and Christian Szegedy.
\newblock Explaining and harnessing adversarial examples.
\newblock In \emph{Proceedings of the 3rd International Conference on Learning
Representations}, 2015.
\bibitem[Goodfellow et~al.(2016)Goodfellow, Bengio, and
Courville]{GoodfellowEtAl2016DeepLearning}
Ian~J. Goodfellow, Yoshua Bengio, and Aaron~C. Courville.
\newblock \emph{Deep Learning}.
\newblock Adaptive Computation and Machine Learning. MIT Press, 2016.
\newblock ISBN 978-0-262-03561-3.
\newblock URL \url{https://www.deeplearningbook.org/}.
\bibitem[Graves et~al.(2013)Graves, rahman Mohamed, and
Hinton]{graves-etal:2013speech}
Alex Graves, Abdel rahman Mohamed, and Geoffrey~E. Hinton.
\newblock Speech recognition with deep recurrent neural networks.
\newblock In \emph{2013 IEEE International Conference on Acoustics, Speech and
Signal Processing}, pages 6645--6649, 2013.
\bibitem[Graves et~al.(2014)Graves, Wayne, and
Danihelka]{graves-etal:2014neural}
Alex Graves, Greg Wayne, and Ivo Danihelka.
\newblock Neural {Turing} machines.
\newblock \emph{arXiv preprint arXiv:1410.5401}, 2014.
\bibitem[Greenberg(2025)]{greenberg-etal:2025demystifying}
Or~Greenberg.
\newblock Demystifying flux architecture.
\newblock \emph{arXiv preprint arXiv:2507.09595}, 2025.
\bibitem[Gu et~al.(2018)Gu, Bradbury, Xiong, Li, and Socher]{gu-etal:2018non}
Jiatao Gu, James Bradbury, Caiming Xiong, Victor~O.K. Li, and Richard Socher.
\newblock Non-autoregressive neural machine translation.
\newblock In \emph{Proceedings of International Conference on Learning
Representations}, 2018.
\bibitem[Gu et~al.(2019)Gu, Wang, and Zhao]{gu-etal:2019levenshtein}
Jiatao Gu, Changhan Wang, and Junbo Zhao.
\newblock Levenshtein transformer.
\newblock \emph{Advances in neural information processing systems}, 32, 2019.
\bibitem[Gunasekar et~al.(2023)Gunasekar, Zhang, Aneja, Mendes, Giorno, Gopi,
Javaheripi, Kauffmann, de~Rosa, Saarikivi, Salim, Shah, Behl, Wang, Bubeck,
Eldan, Kalai, Lee, and Li]{gunasekar-etal:2023textbooks}
Suriya Gunasekar, Yi~Zhang, Jyoti Aneja, Caio C{\'{e}}sar~Teodoro Mendes,
Allie~Del Giorno, Sivakanth Gopi, Mojan Javaheripi, Piero Kauffmann, Gustavo
de~Rosa, Olli Saarikivi, Adil Salim, Shital Shah, Harkirat~Singh Behl, Xin
Wang, S{\'{e}}bastien Bubeck, Ronen Eldan, Adam~Tauman Kalai, Yin~Tat Lee,
and Yuanzhi Li.
\newblock Textbooks are all you need.
\newblock \emph{arXiv preprint arXiv:2306.11644}, 2023.
\bibitem[Harris(1954)]{harris:1954distributional}
Zellig~S. Harris.
\newblock Distributional structure.
\newblock \emph{Word}, 10\penalty0 (2--3):\penalty0 146--162, 1954.
\bibitem[Hastie et~al.(2009)Hastie, Tibshirani, and Friedman]{Hastie-etal:2009}
Trevor Hastie, Robert Tibshirani, and Jerome Friedman.
\newblock \emph{The Elements of Statistical Learning}.
\newblock Springer, 2009.
\bibitem[He et~al.(2016)He, Zhang, Ren, and Sun]{HeEtAl2016ResNet}
Kaiming He, Xiangyu Zhang, Shaoqing Ren, and Jian Sun.
\newblock Deep residual learning for image recognition.
\newblock In \emph{Proceedings of the IEEE Conference on Computer Vision and
Pattern Recognition}, pages 770--778, 2016.
\newblock \doi{10.1109/CVPR.2016.90}.
\bibitem[He et~al.(2022)He, Chen, Xie, Li, Doll{\'a}r, and
Girshick]{HeEtAl2022MAE}
Kaiming He, Xinlei Chen, Saining Xie, Yanghao Li, Piotr Doll{\'a}r, and Ross~B.
Girshick.
\newblock Masked autoencoders are scalable vision learners.
\newblock In \emph{Proceedings of the IEEE/CVF Conference on Computer Vision
and Pattern Recognition}, pages 15979--15988, 2022.
\newblock \doi{10.1109/CVPR52688.2022.01553}.
\bibitem[Hestness et~al.(2017)Hestness, Narang, Ardalani, Diamos, Jun,
Kianinejad, Patwary, Yang, and Zhou]{hestness-etal:2017deep}
Joel Hestness, Sharan Narang, Newsha Ardalani, Gregory Diamos, Heewoo Jun,
Hassan Kianinejad, Md~Mostofa~Ali Patwary, Yang Yang, and Yanqi Zhou.
\newblock Deep learning scaling is predictable, empirically.
\newblock \emph{arXiv preprint arXiv:1712.00409}, 2017.
\bibitem[Hinton(2018)]{Hinton:2018RMSProp}
Geoffrey~E. Hinton.
\newblock Neural networks for machine learning, lecture 6, 2018.
\newblock URL
\url{http://www.cs.toronto.edu/~tijmen/csc321/slides/lecture_slides_lec6.pdf}.
\bibitem[Hinton et~al.(2012)Hinton, Srivastava, Krizhevsky, Sutskever, and
Salakhutdinov]{hinton-etal:2012improving}
Geoffrey~E. Hinton, Nitish Srivastava, Alex Krizhevsky, Ilya Sutskever, and
Ruslan~R. Salakhutdinov.
\newblock Improving neural networks by preventing co-adaptation of feature
detectors.
\newblock \emph{arXiv preprint arXiv:1207.0580}, 2012.
\bibitem[Ho et~al.(2020)Ho, Jain, and Abbeel]{ho-eatl:2020denoising}
Jonathan Ho, Ajay Jain, and Pieter Abbeel.
\newblock Denoising diffusion probabilistic models.
\newblock \emph{Advances in neural information processing systems},
33:\penalty0 6840--6851, 2020.
\bibitem[Hoffmann et~al.(2022)Hoffmann, Borgeaud, Mensch, Buchatskaya, Cai,
Rutherford, de~Las~Casas, Hendricks, Welbl, Clark, Hennigan, Noland,
Millican, van~den Driessche, Damoc, Guy, Osindero, Simonyan, Elsen, Rae,
Vinyals, and Sifre]{hoffmann-etal:2022training}
Jordan Hoffmann, Sebastian Borgeaud, Arthur Mensch, Elena Buchatskaya, Trevor
Cai, Eliza Rutherford, Diego de~Las~Casas, Lisa~Anne Hendricks, Johannes
Welbl, Aidan Clark, Tom Hennigan, Eric Noland, Katie Millican, George van~den
Driessche, Bogdan Damoc, Aurelia Guy, Simon Osindero, Karen Simonyan, Erich
Elsen, Jack~W. Rae, Oriol Vinyals, and Laurent Sifre.
\newblock Training compute-optimal large language models.
\newblock \emph{arXiv preprint arXiv:2203.15556}, 2022.
\bibitem[Holmstr{\"o}m and Koistinen(1992)]{Holmstrom-Koistinen:1992noise}
Lasse Holmstr{\"o}m and Petri Koistinen.
\newblock Using additive noise in back-propagation training.
\newblock \emph{IEEE Transactions on Neural Networks}, 3\penalty0 (1):\penalty0
24--38, 1992.
\bibitem[Honovich et~al.(2023)Honovich, Scialom, Levy, and
Schick]{honovich-etal:2023unnatural}
Or~Honovich, Thomas Scialom, Omer Levy, and Timo Schick.
\newblock Unnatural instructions: Tuning language models with (almost) no human
labor.
\newblock In \emph{Proceedings of the 61st Annual Meeting of the Association
for Computational Linguistics (Volume 1: Long Papers)}, pages 14409--14428,
2023.
\bibitem[Hoogeboom et~al.(2021)Hoogeboom, Nielsen, Jaini, Forr{\'e}, and
Welling]{hoogeboom-etal:2021argmax}
Emiel Hoogeboom, Didrik Nielsen, Priyank Jaini, Patrick Forr{\'e}, and Max
Welling.
\newblock Argmax flows and multinomial diffusion: Learning categorical
distributions.
\newblock \emph{Advances in neural information processing systems},
34:\penalty0 12454--12465, 2021.
\bibitem[Houlsby et~al.(2019)Houlsby, Giurgiu, Jastrzebski, Morrone,
de~Laroussilhe, Gesmundo, Attariyan, and Gelly]{houlsby-etal:2019adapter}
Neil Houlsby, Andrei Giurgiu, Stanislaw Jastrzebski, Bruna Morrone, Quentin
de~Laroussilhe, Andrea Gesmundo, Mona Attariyan, and Sylvain Gelly.
\newblock Parameter-efficient transfer learning for nlp.
\newblock In \emph{International Conference on Machine Learning (ICML)}, 2019.
\bibitem[Hu et~al.(2021)Hu, Shen, Wallis, Allen-Zhu, Li, Wang, Wang, and
Chen]{hu-etal:2021lora}
Edward~J Hu, Yelong Shen, Phillip Wallis, Zeyuan Allen-Zhu, Yuanzhi Li, Shean
Wang, Lu~Wang, and Weizhu Chen.
\newblock Lora: Low-rank adaptation of large language models.
\newblock In \emph{International Conference on Learning Representations
(ICLR)}, 2021.
\bibitem[Huang et~al.(2019)Huang, Cheng, Bapna, Firat, Chen, Chen, Lee, Ngiam,
Le, Wu, and Chen]{huang-etal:2019gpipe}
Yanping Huang, Youlong Cheng, Ankur Bapna, Orhan Firat, Mia~Xu Chen, Dehao
Chen, HyoukJoong Lee, Jiquan Ngiam, Quoc~V Le, Yonghui Wu, and Zhifeng Chen.
\newblock Gpipe: Efficient training of giant neural networks using pipeline
parallelism.
\newblock \emph{Advances in neural information processing systems}, 32, 2019.
\bibitem[Jang et~al.(2017)Jang, Gu, and Poole]{jang-etal:2017categorical}
Eric Jang, Shixiang Gu, and Ben Poole.
\newblock Categorical reparameterization with gumbel-softmax.
\newblock In \emph{International Conference on Learning Representations}, 2017.
\bibitem[Jolliffe(2002)]{Jolliffe2002PCA}
Ian~T. Jolliffe.
\newblock \emph{Principal Component Analysis}.
\newblock Springer Series in Statistics. Springer, New York, 2 edition, 2002.
\newblock ISBN 978-0-387-95442-4.
\newblock \doi{10.1007/b98835}.
\bibitem[Joshi et~al.(2017)Joshi, Choi, Weld, and
Zettlemoyer]{joshi-etal:2017triviaqa}
Mandar Joshi, Eunsol Choi, Daniel~S Weld, and Luke Zettlemoyer.
\newblock Triviaqa: A large scale distantly supervised challenge dataset for
reading comprehension.
\newblock In \emph{Proceedings of the 55th Annual Meeting of the Association
for Computational Linguistics (Volume 1: Long Papers)}, pages 1601--1611,
2017.
\bibitem[Joshi et~al.(2020)Joshi, Chen, Liu, Weld, Zettlemoyer, and
Levy]{joshi-etal:2020spanbert}
Mandar Joshi, Danqi Chen, Yinhan Liu, Daniel~S Weld, Luke Zettlemoyer, and Omer
Levy.
\newblock Spanbert: Improving pre-training by representing and predicting
spans.
\newblock \emph{Transactions of the association for computational linguistics},
8:\penalty0 64--77, 2020.
\bibitem[Kahneman(2011)]{kahneman:2011thinking}
Daniel Kahneman.
\newblock \emph{Thinking, fast and slow}.
\newblock macmillan, 2011.
\bibitem[Kaplan et~al.(2020)Kaplan, McCandlish, Henighan, Brown, Chess, Child,
Gray, Radford, Wu, and Amodei]{kaplan-etal:2020scaling}
Jared Kaplan, Sam McCandlish, Tom Henighan, Tom~B Brown, Benjamin Chess, Rewon
Child, Scott Gray, Alec Radford, Jeffrey Wu, and Dario Amodei.
\newblock Scaling laws for neural language models.
\newblock \emph{arXiv preprint arXiv:2001.08361}, 2020.
\bibitem[Kelly and Stone(1975)]{Kelly-Stone:1975senses}
Edward~F. Kelly and Philip~J. Stone.
\newblock \emph{Computer Recognition of English Word Senses}.
\newblock American Elsevier Pub, 1975.
\bibitem[Kingma and Ba(2015)]{KingmaBa2015Adam}
Diederik~P. Kingma and Jimmy Ba.
\newblock Adam: A method for stochastic optimization.
\newblock In \emph{Proceedings of the 3rd International Conference on Learning
Representations}, 2015.
\newblock URL \url{https://arxiv.org/abs/1412.6980}.
\bibitem[Koh et~al.(2025)Koh, Jhang, Kim, Lee, and
Jung]{koh-etal:2025conditional}
Hyukhun Koh, Minha Jhang, Dohyung Kim, Sangmook Lee, and Kyomin Jung.
\newblock Conditional [mask] discrete diffusion language model.
\newblock In \emph{Proceedings of the 2025 Conference on Empirical Methods in
Natural Language Processing}, pages 8910--8934, 2025.
\bibitem[Kojima et~al.(2022)Kojima, Gu, Reid, Matsuo, and
Iwasawa]{kojima-etal:2022large}
Takeshi Kojima, Shixiang~Shane Gu, Machel Reid, Yutaka Matsuo, and Yusuke
Iwasawa.
\newblock Large language models are zero-shot reasoners.
\newblock \emph{Advances in neural information processing systems},
35:\penalty0 22199--22213, 2022.
\bibitem[Lee et~al.(2023)Lee, Phatale, Mansoor, Lu, Mesnard, Ferret, Bishop,
Hall, Carbune, and Rastogi]{lee-etal:2023rlaif}
Harrison Lee, Samrat Phatale, Hassan Mansoor, Kellie~Ren Lu, Thomas Mesnard,
Johan Ferret, Colton Bishop, Ethan Hall, Victor Carbune, and Abhinav Rastogi.
\newblock Rlaif: Scaling reinforcement learning from human feedback with ai
feedback.
\newblock \emph{arXiv preprint arXiv:2309.00267}, 2023.
\bibitem[Lee et~al.(2018)Lee, Mansimov, and Cho]{lee-etal:2018deterministic}
Jason Lee, Elman Mansimov, and Kyunghyun Cho.
\newblock Deterministic non-autoregressive neural sequence modeling by
iterative refinement.
\newblock In \emph{Proceedings of the 2018 Conference on Empirical Methods in
Natural Language Processing}, pages 1173--1182, 2018.
\bibitem[Lester et~al.(2021{\natexlab{a}})Lester, Al-Rfou, and
Constant]{lester-etal:2021power}
Brian Lester, Rami Al-Rfou, and Noah Constant.
\newblock The power of scale for parameter-efficient prompt tuning.
\newblock In \emph{Proceedings of the 2021 Conference on Empirical Methods in
Natural Language Processing}, pages 3045--3059, 2021{\natexlab{a}}.
\bibitem[Lester et~al.(2021{\natexlab{b}})Lester, Al-Rfou, and
Constant]{lester-etal:2021prompt}
Brian Lester, Rami Al-Rfou, and Noah Constant.
\newblock The power of scale for parameter-efficient prompt tuning.
\newblock In \emph{Proceedings of the 2021 Conference on Empirical Methods in
Natural Language Processing (EMNLP)}, 2021{\natexlab{b}}.
\bibitem[Lewis et~al.(2020{\natexlab{a}})Lewis, Liu, Goyal, Ghazvininejad,
Mohamed, Levy, Stoyanov, and Zettlemoyer]{LewisEtAl2020BART}
Mike Lewis, Yinhan Liu, Naman Goyal, Marjan Ghazvininejad, Abdelrahman Mohamed,
Omer Levy, Veselin Stoyanov, and Luke Zettlemoyer.
\newblock {BART}: Denoising sequence-to-sequence pre-training for natural
language generation, translation, and comprehension.
\newblock In \emph{Proceedings of the 58th Annual Meeting of the Association
for Computational Linguistics}, pages 7871--7880. Association for
Computational Linguistics, 2020{\natexlab{a}}.
\newblock \doi{10.18653/v1/2020.acl-main.703}.
\bibitem[Lewis et~al.(2020{\natexlab{b}})Lewis, Liu, Goyal, Ghazvininejad,
Mohamed, Levy, Stoyanov, and Zettlemoyer]{lewis-etal:2020bart}
Mike Lewis, Yinhan Liu, Naman Goyal, Marjan Ghazvininejad, Abdelrahman Mohamed,
Omer Levy, Veselin Stoyanov, and Luke Zettlemoyer.
\newblock Bart: Denoising sequence-to-sequence pre-training for natural
language generation, translation, and comprehension.
\newblock In \emph{Proceedings of the 58th Annual Meeting of the Association
for Computational Linguistics}, pages 7871--7880, 2020{\natexlab{b}}.
\bibitem[Li et~al.(2018)Li, Farkhoor, Liu, and Yosinski]{li-etal:2018intrinsic}
Chunyuan Li, Heerad Farkhoor, Rosanne Liu, and Jason Yosinski.
\newblock Measuring the intrinsic dimension of objective landscapes.
\newblock In \emph{International Conference on Learning Representations
(ICLR)}, 2018.
\bibitem[Li et~al.(2025)Li, Dong, Zang, Cao, Wang, and Lin]{li-etal:2025fixed}
Jinsong Li, Xiaoyi Dong, Yuhang Zang, Yuhang Cao, Jiaqi Wang, and Dahua Lin.
\newblock Beyond fixed: Training-free variable-length denoising for diffusion
large language models, 2025.
\bibitem[Li et~al.(2022)Li, Thickstun, Gulrajani, Liang, and
Hashimoto]{li-etal:2022diffusion}
Xiang Li, John Thickstun, Ishaan Gulrajani, Percy~S Liang, and Tatsunori~B
Hashimoto.
\newblock Diffusion-lm improves controllable text generation.
\newblock \emph{Advances in neural information processing systems},
35:\penalty0 4328--4343, 2022.
\bibitem[Li and Liang(2021)]{li-liang:2021prefix}
Xiang~Lisa Li and Percy Liang.
\newblock Prefix-tuning: Optimizing continuous prompts for generation.
\newblock In \emph{Proceedings of the 59th Annual Meeting of the Association
for Computational Linguistics and the 11th International Joint Conference on
Natural Language Processing (Volume 1: Long Papers)}, pages 4582--4597, 2021.
\bibitem[Lightman et~al.(2024)Lightman, Kosaraju, Burda, Edwards, Baker, Lee,
Leike, Schulman, Sutskever, and Cobbe]{lightman-etal:2024lets}
Hunter Lightman, Vineet Kosaraju, Yuri Burda, Harrison Edwards, Bowen Baker,
Teddy Lee, Jan Leike, John Schulman, Ilya Sutskever, and Karl Cobbe.
\newblock Let's verify step by step.
\newblock In \emph{The Twelfth International Conference on Learning
Representations}, 2024.
\bibitem[Liu et~al.(2024{\natexlab{a}})Liu, Feng, Xue, Wang, Wu, Lu, Zhao,
Deng, Zhang, Ruan, et~al.]{liu2024deepseek}
Aixin Liu, Bei Feng, Bing Xue, Bingxuan Wang, Bochao Wu, Chengda Lu, Chenggang
Zhao, Chengqi Deng, Chenyu Zhang, Chong Ruan, et~al.
\newblock Deepseek-v3 technical report.
\newblock \emph{arXiv preprint arXiv:2412.19437}, 2024{\natexlab{a}}.
\bibitem[Liu et~al.(2023)Liu, Yuan, Fu, Jiang, Hayashi, and
Neubig]{liu-etal:2023pre}
Pengfei Liu, Weizhe Yuan, Jinlan Fu, Zhengbao Jiang, Hiroaki Hayashi, and
Graham Neubig.
\newblock Pre-train, prompt, and predict: A systematic survey of prompting
methods in natural language processing.
\newblock \emph{ACM Computing Surveys}, 55\penalty0 (9):\penalty0 1--35, 2023.
\bibitem[Liu et~al.(2024{\natexlab{b}})Liu, Wang, Yin, Molchanov, Wang, Cheng,
and Chen]{liu-etal:2024dora}
Shih-Yang Liu, Chien-Yi Wang, Hongxu Yin, Pavlo Molchanov, Yu-Chiang~Frank
Wang, Kwang-Ting Cheng, and Min-Hung Chen.
\newblock Dora: Weight-decomposed low-rank adaptation.
\newblock In \emph{Proceedings of the 41st International Conference on Machine
Learning (ICML)}, 2024{\natexlab{b}}.
\bibitem[Longpre et~al.(2023)Longpre, Hou, Vu, Webson, Chung, Tay, Zhou, Le,
Zoph, Wei, and Roberts]{longpre-etal:2023flan}
Shayne Longpre, Le~Hou, Tu~Vu, Albert Webson, Hyung~Won Chung, Yi~Tay, Denny
Zhou, Quoc~V. Le, Barret Zoph, Jason Wei, and Adam Roberts.
\newblock The flan collection: Designing data and methods for effective
instruction tuning.
\newblock In \emph{International Conference on Machine Learning}, pages
22631--22648. PMLR, 2023.
\bibitem[Lou et~al.(2024)Lou, Meng, and Ermon]{lou-etal:2024discrete}
Aaron Lou, Chenlin Meng, and Stefano Ermon.
\newblock Discrete diffusion modeling by estimating the ratios of the data
distribution.
\newblock In \emph{Proceedings of the 41st International Conference on Machine
Learning}, pages 32819--32848, 2024.
\bibitem[Luong et~al.(2015)Luong, Pham, and Manning]{luong-etal:2015effective}
Minh-Thang Luong, Hieu Pham, and Christopher~D. Manning.
\newblock Effective approaches to attention-based neural machine translation.
\newblock In \emph{Proceedings of the 2015 Conference on Empirical Methods in
Natural Language Processing}, pages 1412--1421, 2015.
\bibitem[Maddison et~al.(2017)Maddison, Mnih, and Teh]{maddison-etal:2017the}
Chris~J. Maddison, Andriy Mnih, and Yee~Whye Teh.
\newblock The concrete distribution: A continuous relaxation of discrete random
variables.
\newblock In \emph{International Conference on Learning Representations}, 2017.
\bibitem[Manning et~al.(2008)Manning, Raghavan, and
Sch{"u}tze]{ManningEtAl2008IR}
Christopher~D. Manning, Prabhakar Raghavan, and Hinrich Sch{"u}tze.
\newblock \emph{Introduction to Information Retrieval}.
\newblock Cambridge University Press, 2008.
\newblock ISBN 978-0-521-86571-5.
\newblock URL \url{https://nlp.stanford.edu/IR-book/}.
\bibitem[McClave and Sincich(2006)]{McClave-and-Sincich-2006}
James~T. McClave and Terry Sincich.
\newblock \emph{Statistics}.
\newblock Prentice Hall, 10 edition, 2006.
\bibitem[Micikevicius et~al.(2018)Micikevicius, Narang, Alben, Diamos, Elsen,
Garcia, Ginsburg, Houston, Kuchaiev, Venkatesh, and
Wu]{micikevicius-etal:2018mixed}
Paulius Micikevicius, Sharan Narang, Jonah Alben, Gregory Diamos, Erich Elsen,
David Garcia, Boris Ginsburg, Michael Houston, Oleksii Kuchaiev, Ganesh
Venkatesh, and Hao Wu.
\newblock Mixed precision training.
\newblock In \emph{Proceedings of International Conference on Learning
Representations}, 2018.
\bibitem[Mikolov et~al.(2013{\natexlab{a}})Mikolov, Chen, Corrado, and
Dean]{MikolovEtAl2013Word2Vec}
Tom{\'a}s Mikolov, Kai Chen, Greg Corrado, and Jeffrey Dean.
\newblock Efficient estimation of word representations in vector space.
\newblock In \emph{Proceedings of the 1st International Conference on Learning
Representations, Workshop Track}, 2013{\natexlab{a}}.
\newblock URL \url{https://arxiv.org/abs/1301.3781}.
\bibitem[Mikolov et~al.(2013{\natexlab{b}})Mikolov, Chen, Corrado, and
Dean]{mikolov-etal:2013efficient}
Tomas Mikolov, Kai Chen, Greg Corrado, and Jeffrey Dean.
\newblock Efficient estimation of word representations in vector space.
\newblock In \emph{Proceedings of the International Conference on Learning
Representations (ICLR 2013)}, 2013{\natexlab{b}}.
\bibitem[Mikolov et~al.(2013{\natexlab{c}})Mikolov, Sutskever, Chen, Corrado,
and Dean]{mikolov-etal:2013distributed}
Tomas Mikolov, Ilya Sutskever, Kai Chen, Greg Corrado, and Jeffrey Dean.
\newblock Distributed representations of words and phrases and their
compositionality.
\newblock In \emph{Proceedings of the 26th International Conference on Neural
Information Processing Systems - Volume 2}, pages 3111--3119,
2013{\natexlab{c}}.
\bibitem[Mishra et~al.(2022)Mishra, Khashabi, Baral, and
Hajishirzi]{mishra-etal:2022cross}
Swaroop Mishra, Daniel Khashabi, Chitta Baral, and Hannaneh Hajishirzi.
\newblock Cross-task generalization via natural language crowdsourcing
instructions.
\newblock In \emph{Proceedings of the 60th Annual Meeting of the Association
for Computational Linguistics (Volume 1: Long Papers)}, pages 3470--3487,
2022.
\bibitem[{Mistral AI}(2025)]{mistral2025mistral3}
{Mistral AI}.
\newblock Introducing mistral 3.
\newblock \url{https://mistral.ai/news/mistral-3}, 2025.
\newblock Accessed: 2026-07-28.
\bibitem[Mitchell(1997)]{Mitchell:1997}
Tom~M. Mitchell.
\newblock \emph{Machine Learning}.
\newblock McGraw-Hill Education, 1997.
\bibitem[Monsefi et~al.(2025)Monsefi, Bhendawade, Ciosici, Culver, Zhang, and
Belousova]{monsefi-etal:2025fs}
Amin~Karimi Monsefi, Nikhil Bhendawade, Manuel~Rafael Ciosici, Dominic Culver,
Yizhe Zhang, and Irina Belousova.
\newblock Fs-dfm: Fast and accurate long text generation with few-step
diffusion language models.
\newblock \emph{arXiv preprint arXiv:2509.20624}, 2025.
\bibitem[Narayanan et~al.(2021)Narayanan, Shoeybi, Casper, LeGresley, Patwary,
Korthikanti, Vainbrand, Kashinkunti, Bernauer, Catanzaro, Phanishayee, and
Zaharia]{narayanan-etal:2021efficient}
Deepak Narayanan, Mohammad Shoeybi, Jared Casper, Patrick LeGresley, Mostofa
Patwary, Vijay Korthikanti, Dmitri Vainbrand, Prethvi Kashinkunti, Julie
Bernauer, Bryan Catanzaro, Amar Phanishayee, and Matei Zaharia.
\newblock Efficient large-scale language model training on gpu clusters using
megatron-lm.
\newblock In \emph{Proceedings of the International Conference for High
Performance Computing, Networking, Storage and Analysis}, pages 1--15, 2021.
\bibitem[Neelakantan et~al.(2015)Neelakantan, Vilnis, Le, Sutskever, Kaiser,
Kurach, and Martens]{neelakantan-etal:2015adding}
Arvind Neelakantan, Luke Vilnis, Quoc~V. Le, Ilya Sutskever, Lukasz Kaiser,
Karol Kurach, and James Martens.
\newblock Adding gradient noise improves learning for very deep networks.
\newblock \emph{arXiv preprint arXiv:1511.06807}, 2015.
\bibitem[Nie et~al.(2025)Nie, Zhu, You, Zhang, Ou, Hu, Zhou, Lin, Wen, and
Li]{nie-etal:2025large}
Shen Nie, Fengqi Zhu, Zebin You, Xiaolu Zhang, Jingyang Ou, Jun Hu, Jun Zhou,
Yankai Lin, Ji-Rong Wen, and Chongxuan Li.
\newblock Large language diffusion models.
\newblock \emph{arXiv preprint arXiv:2502.09992}, 2025.
\bibitem[Ouyang et~al.(2022)Ouyang, Wu, Jiang, Almeida, Wainwright, Mishkin,
Zhang, Agarwal, Slama, Ray, Schulman, Hilton, Kelton, Miller, Simens, Askell,
Welinder, Christiano, Leike, and Lowe]{ouyang-etal:2022training}
Long Ouyang, Jeffrey Wu, Xu~Jiang, Diogo Almeida, Carroll~L. Wainwright, Pamela
Mishkin, Chong Zhang, Sandhini Agarwal, Katarina Slama, Alex Ray, John
Schulman, Jacob Hilton, Fraser Kelton, Luke Miller, Maddie Simens, Amanda
Askell, Peter Welinder, Paul~F. Christiano, Jan Leike, and Ryan Lowe.
\newblock Training language models to follow instructions with human feedback.
\newblock \emph{Advances in Neural Information Processing Systems},
35:\penalty0 27730--27744, 2022.
\bibitem[Pan and Yang(2010)]{PanYang2010Transfer}
Sinno~Jialin Pan and Qiang Yang.
\newblock A survey on transfer learning.
\newblock \emph{IEEE Transactions on Knowledge and Data Engineering},
22\penalty0 (10):\penalty0 1345--1359, 2010.
\newblock \doi{10.1109/TKDE.2009.191}.
\bibitem[Papineni et~al.(2002)Papineni, Roukos, Ward, and
Zhu]{PapineniEtAl2002BLEU}
Kishore Papineni, Salim Roukos, Todd Ward, and Wei{-}Jing Zhu.
\newblock {BLEU}: A method for automatic evaluation of machine translation.
\newblock In \emph{Proceedings of the 40th Annual Meeting of the Association
for Computational Linguistics}, pages 311--318. Association for Computational
Linguistics, 2002.
\newblock \doi{10.3115/1073083.1073135}.
\bibitem[Peebles and Xie(2023)]{peebles-and-xie:2023scalable}
William Peebles and Saining Xie.
\newblock Scalable diffusion models with transformers.
\newblock In \emph{Proceedings of the IEEE/CVF international conference on
computer vision}, pages 4195--4205, 2023.
\bibitem[Peirce(1931)]{peirce1931:collected}
Charles~Sanders Peirce.
\newblock \emph{Collected Papers of Charles Sanders Peirce}, volume 1-6.
\newblock Harvard University Press, 1931.
\bibitem[Penedo et~al.(2023)Penedo, Malartic, Hesslow, Cojocaru, Cappelli,
Alobeidli, Pannier, Almazrouei, and Launay]{penedo-etal:2023refinedweb}
Guilherme Penedo, Quentin Malartic, Daniel Hesslow, Ruxandra Cojocaru,
Alessandro Cappelli, Hamza Alobeidli, Baptiste Pannier, Ebtesam Almazrouei,
and Julien Launay.
\newblock The refinedweb dataset for falcon llm: outperforming curated corpora
with web data, and web data only.
\newblock \emph{arXiv preprint arXiv:2306.01116}, 2023.
\bibitem[Pennington et~al.(2014)Pennington, Socher, and
Manning]{PenningtonEtAl2014GloVe}
Jeffrey Pennington, Richard Socher, and Christopher~D. Manning.
\newblock {GloVe}: Global vectors for word representation.
\newblock In \emph{Proceedings of the 2014 Conference on Empirical Methods in
Natural Language Processing}, pages 1532--1543. Association for Computational
Linguistics, 2014.
\newblock \doi{10.3115/v1/D14-1162}.
\bibitem[Plaut et~al.(1986)Plaut, Nowlan, and
Hinton]{plaut-etal:1986experiments}
David~C. Plaut, Steven~J. Nowlan, and Geoffrey~E. Hinton.
\newblock Experiments on learning by back propagation.
\newblock Technical report, Carnegie-Mellon University, 1986.
\bibitem[Prechelt(1998)]{prechelt1998early}
Lutz Prechelt.
\newblock Early stopping---but when?
\newblock In \emph{Neural Networks: Tricks of the Trade}, pages 55--69.
Springer, 1998.
\bibitem[Radford et~al.(2018)Radford, Narasimhan, Salimans, and
Sutskever]{radford-etal:2018improving}
Alec Radford, Karthik Narasimhan, Tim Salimans, and Ilya Sutskever.
\newblock Improving language understanding by generative pre-training.
\newblock \emph{OpenAI Technical Report}, 2018.
\bibitem[Radford et~al.(2019)Radford, Wu, Child, Luan, Amodei, and
Sutskever]{radford-etal:2019language}
Alec Radford, Jeffrey Wu, Rewon Child, David Luan, Dario Amodei, and Ilya
Sutskever.
\newblock Language models are unsupervised multitask learners.
\newblock \emph{OpenAI blog}, 1\penalty0 (8), 2019.
\bibitem[Rafailov et~al.(2024)Rafailov, Sharma, Mitchell, Manning, Ermon, and
Finn]{rafailov-etal:2024direct}
Rafael Rafailov, Archit Sharma, Eric Mitchell, Christopher~D Manning, Stefano
Ermon, and Chelsea Finn.
\newblock Direct preference optimization: Your language model is secretly a
reward model.
\newblock \emph{Advances in Neural Information Processing Systems}, 36, 2024.
\bibitem[Raffel et~al.(2020{\natexlab{a}})Raffel, Shazeer, Roberts, Lee,
Narang, Matena, Zhou, Li, and Liu]{RaffelEtAl2020T5}
Colin Raffel, Noam Shazeer, Adam Roberts, Katherine Lee, Sharan Narang, Michael
Matena, Yanqi Zhou, Wei Li, and Peter~J. Liu.
\newblock Exploring the limits of transfer learning with a unified text-to-text
transformer.
\newblock \emph{Journal of Machine Learning Research}, 21\penalty0
(140):\penalty0 1--67, 2020{\natexlab{a}}.
\newblock URL \url{https://jmlr.org/papers/v21/20-074.html}.
\bibitem[Raffel et~al.(2020{\natexlab{b}})Raffel, Shazeer, Roberts, Lee,
Narang, Matena, Zhou, Li, and Liu]{raffel-etal:2020exploring}
Colin Raffel, Noam Shazeer, Adam Roberts, Katherine Lee, Sharan Narang, Michael
Matena, Yanqi Zhou, Wei Li, and Peter~J. Liu.
\newblock Exploring the limits of transfer learning with a unified text-to-text
transformer.
\newblock \emph{Journal of Machine Learning Research}, 21\penalty0
(140):\penalty0 1--67, 2020{\natexlab{b}}.
\bibitem[Reid et~al.(2022)Reid, Hellendoorn, and
Neubig]{reid-etal:2022diffuser}
Machel Reid, Vincent~J Hellendoorn, and Graham Neubig.
\newblock Diffuser: Discrete diffusion via edit-based reconstruction.
\newblock \emph{arXiv preprint arXiv:2210.16886}, 2022.
\bibitem[Rombach et~al.(2022)Rombach, Blattmann, Lorenz, Esser, and
Ommer]{rombach-etal:2022high}
Robin Rombach, Andreas Blattmann, Dominik Lorenz, Patrick Esser, and Bj{\"o}rn
Ommer.
\newblock High-resolution image synthesis with latent diffusion models.
\newblock In \emph{Proceedings of the IEEE/CVF conference on computer vision
and pattern recognition}, pages 10684--10695, 2022.
\bibitem[Rosenblatt(1958)]{Rosenblatt1958Perceptron}
Frank Rosenblatt.
\newblock The perceptron: A probabilistic model for information storage and
organization in the brain.
\newblock \emph{Psychological Review}, 65\penalty0 (6):\penalty0 386--408,
1958.
\newblock \doi{10.1037/h0042519}.
\bibitem[Rumelhart et~al.(1986)Rumelhart, Hinton, and
Williams]{RumelhartEtAl1986Backprop}
David~E. Rumelhart, Geoffrey~E. Hinton, and Ronald~J. Williams.
\newblock Learning representations by back-propagating errors.
\newblock \emph{Nature}, 323\penalty0 (6088):\penalty0 533--536, 1986.
\newblock \doi{10.1038/323533a0}.
\bibitem[R{\"u}tte et~al.(2025)R{\"u}tte, Fluri, Ding, Orvieto, Sch{\"o}lkopf,
and Hofmann]{rutte-etal:2025generalized}
Dimitri~von R{\"u}tte, Janis Fluri, Yuhui Ding, Antonio Orvieto, Bernhard
Sch{\"o}lkopf, and Thomas Hofmann.
\newblock Generalized interpolating discrete diffusion.
\newblock In \emph{Forty-second International Conference on Machine Learning},
2025.
\bibitem[Sahoo et~al.(2024)Sahoo, Arriola, Schiff, Gokaslan, Marroquin, Chiu,
Rush, and Kuleshov]{sahoo-etal:2024simple}
Subham Sahoo, Marianne Arriola, Yair Schiff, Aaron Gokaslan, Edgar Marroquin,
Justin Chiu, Alexander Rush, and Volodymyr Kuleshov.
\newblock Simple and effective masked diffusion language models.
\newblock \emph{Advances in Neural Information Processing Systems},
37:\penalty0 130136--130184, 2024.
\bibitem[Saini et~al.(2025)Saini, Gupta, and Bovik]{saini-etal:2025rectified}
Shreshth Saini, Shashank Gupta, and Alan~C Bovik.
\newblock Rectified-cfg++ for flow based models.
\newblock \emph{arXiv preprint arXiv:2510.07631}, 2025.
\bibitem[Salimans and Ho(2022)]{salimans-and-ho:2022progressive}
Tim Salimans and Jonathan Ho.
\newblock Progressive distillation for fast sampling of diffusion models.
\newblock In \emph{International Conference on Learning Representations}, 2022.
\bibitem[Sanh et~al.(2022)Sanh, Webson, Raffel, Bach, Sutawika, Alyafeai,
Chaffin, Stiegler, Raja, Dey, Bari, Xu, Thakker, Sharma, Szczechla, Kim,
Chhablani, Nayak, Datta, Chang, Jiang, Wang, Manica, Shen, Yong, Pandey,
Bawden, Wang, Neeraj, Rozen, Sharma, Santilli, Fevry, Fries, Teehan, Scao,
Biderman, Gao, Wolf, and Rush]{sanh-etal:2022multitask}
Victor Sanh, Albert Webson, Colin Raffel, Stephen Bach, Lintang Sutawika, Zaid
Alyafeai, Antoine Chaffin, Arnaud Stiegler, Arun Raja, Manan Dey, M~Saiful
Bari, Canwen Xu, Urmish Thakker, Shanya~Sharma Sharma, Eliza Szczechla,
Taewoon Kim, Gunjan Chhablani, Nihal Nayak, Debajyoti Datta, Jonathan Chang,
Mike Tian-Jian Jiang, Han Wang, Matteo Manica, Sheng Shen, Zheng~Xin Yong,
Harshit Pandey, Rachel Bawden, Thomas Wang, Trishala Neeraj, Jos Rozen,
Abheesht Sharma, Andrea Santilli, Thibault Fevry, Jason~Alan Fries, Ryan
Teehan, Teven~Le Scao, Stella Biderman, Leo Gao, Thomas Wolf, and Alexander~M
Rush.
\newblock Multitask prompted training enables zero-shot task generalization.
\newblock In \emph{Proceedings of International Conference on Learning
Representations}, 2022.
\bibitem[Sennrich et~al.(2016)Sennrich, Haddow, and
Birch]{sennrich-etal:2016improving}
Rico Sennrich, Barry Haddow, and Alexandra Birch.
\newblock Improving neural machine translation models with monolingual data.
\newblock In \emph{Proceedings of the 54th Annual Meeting of the Association
for Computational Linguistics}, pages 86--96, 2016.
\bibitem[Shaul et~al.(2025)Shaul, Gat, Havasi, Severo, Sriram, Holderrieth,
Karrer, Lipman, and Chen]{shaul-etal:2025flow}
Neta Shaul, Itai Gat, Marton Havasi, Daniel Severo, Anuroop Sriram, Peter
Holderrieth, Brian Karrer, Yaron Lipman, and Ricky T.~Q. Chen.
\newblock Flow matching with general discrete paths: A kinetic-optimal
perspective.
\newblock In \emph{The Thirteenth International Conference on Learning
Representations}, 2025.
\bibitem[Shinn et~al.(2023)Shinn, Cassano, Gopinath, Narasimhan, and
Yao]{shinn-etal:2023reflexion}
Noah Shinn, Federico Cassano, Ashwin Gopinath, Karthik Narasimhan, and Shunyu
Yao.
\newblock Reflexion: Language agents with verbal reinforcement learning.
\newblock \emph{Advances in Neural Information Processing Systems},
36:\penalty0 8634--8652, 2023.
\bibitem[Shorten and Khoshgoftaar(2019)]{shorten-Khoshgoftaar:2019survey}
Connor Shorten and Taghi~M. Khoshgoftaar.
\newblock A survey on image data augmentation for deep learning.
\newblock \emph{Journal of Big Data}, 6\penalty0 (1):\penalty0 1--48, 2019.
\bibitem[Srivastava et~al.(2014)Srivastava, Hinton, Krizhevsky, Sutskever, and
Salakhutdinov]{SrivastavaEtAl2014Dropout}
Nitish Srivastava, Geoffrey~E. Hinton, Alex Krizhevsky, Ilya Sutskever, and
Ruslan Salakhutdinov.
\newblock Dropout: A simple way to prevent neural networks from overfitting.
\newblock \emph{Journal of Machine Learning Research}, 15\penalty0
(1):\penalty0 1929--1958, 2014.
\newblock URL \url{https://jmlr.org/papers/v15/srivastava14a.html}.
\bibitem[Strudel et~al.(2022)Strudel, Tallec, Altch{\'e}, Du, Ganin, Mensch,
Grathwohl, Savinov, Dieleman, Sifre, and Leblond]{strudel-etal:2022self}
Robin Strudel, Corentin Tallec, Florent Altch{\'e}, Yilun Du, Yaroslav Ganin,
Arthur Mensch, Will Grathwohl, Nikolay Savinov, Sander Dieleman, Laurent
Sifre, and R{\' e}mi Leblond.
\newblock Self-conditioned embedding diffusion for text generation.
\newblock \emph{arXiv preprint arXiv:2211.04236}, 2022.
\bibitem[Sutskever et~al.(2013)Sutskever, Martens, Dahl, and
Hinton]{SutskeverEtAl2013Momentum}
Ilya Sutskever, James Martens, George~E. Dahl, and Geoffrey~E. Hinton.
\newblock On the importance of initialization and momentum in deep learning.
\newblock In \emph{Proceedings of the 30th International Conference on Machine
Learning}, volume~28 of \emph{Proceedings of Machine Learning Research},
pages 1139--1147. PMLR, 2013.
\newblock URL \url{https://proceedings.mlr.press/v28/sutskever13.html}.
\bibitem[Sutskever et~al.(2014)Sutskever, Vinyals, and
Le]{SutskeverEtAl2014Seq2Seq}
Ilya Sutskever, Oriol Vinyals, and Quoc~V. Le.
\newblock Sequence to sequence learning with neural networks.
\newblock In \emph{Advances in Neural Information Processing Systems 27}, pages
3104--3112, 2014.
\newblock URL
\url{https://proceedings.neurips.cc/paper/2014/hash/a14ac55a4f27472c5d894ec1c3c743d2-Abstract.html}.
\bibitem[Sutton and Barto(2018)]{SuttonBarto2018RL}
Richard~S. Sutton and Andrew~G. Barto.
\newblock \emph{Reinforcement Learning: An Introduction}.
\newblock MIT Press, 2 edition, 2018.
\newblock ISBN 978-0-262-03924-6.
\newblock URL \url{http://incompleteideas.net/book/the-book-2nd.html}.
\bibitem[Szegedy et~al.(2014)Szegedy, Zaremba, Sutskever, Bruna, Erhan,
Goodfellow, and Fergus]{Szegedy-etal:2014neural}
Christian Szegedy, Wojciech Zaremba, Ilya Sutskever, Joan Bruna, Dumitru Erhan,
Ian Goodfellow, and Rob Fergus.
\newblock Intriguing properties of neural networks.
\newblock In \emph{Proceedings of the 2nd International Conference on Learning
Representations}, 2014.
\bibitem[Szegedy et~al.(2016)Szegedy, Vanhoucke, Ioffe, Shlens, and
Wojna]{SzegedyEtAl2016LabelSmoothing}
Christian Szegedy, Vincent Vanhoucke, Sergey Ioffe, Jonathon Shlens, and
Zbigniew Wojna.
\newblock Rethinking the inception architecture for computer vision.
\newblock In \emph{Proceedings of the IEEE Conference on Computer Vision and
Pattern Recognition}, pages 2818--2826, 2016.
\newblock \doi{10.1109/CVPR.2016.308}.
\bibitem[Taori et~al.(2023)Taori, Gulrajani, Zhang, Dubois, Li, Guestrin,
Liang, and Hashimoto]{taori-etal:2023alpaca}
Rohan Taori, Ishaan Gulrajani, Tianyi Zhang, Yann Dubois, Xuechen Li, Carlos
Guestrin, Percy Liang, and Tatsunori~B. Hashimoto.
\newblock Stanford alpaca: An instruction-following llama model.
\newblock \url{https://github.com/tatsu-lab/stanford_alpaca}, 2023.
\bibitem[Tay et~al.(2023)Tay, Dehghani, Tran, Garcia, Wei, Wang, Chung, Bahri,
Schuster, Zheng, Zhou, Houlsby, and Metzler]{tay-etal:2023ul2}
Yi~Tay, Mostafa Dehghani, Vinh~Q. Tran, Xavier Garcia, Jason Wei, Xuezhi Wang,
Hyung~Won Chung, Dara Bahri, Tal Schuster, Steven Zheng, Denny Zhou, Neil
Houlsby, and Donald Metzler.
\newblock {UL}2: Unifying language learning paradigms.
\newblock In \emph{The Eleventh International Conference on Learning
Representations}, 2023.
\bibitem[Team(2025{\natexlab{a}})]{gemmateam2025gemma3}
Gemma Team.
\newblock Gemma 3 technical report.
\newblock \emph{arXiv preprint arXiv:2503.19786}, 2025{\natexlab{a}}.
\bibitem[Team(2025{\natexlab{b}})]{yang2025qwen3}
Qwen Team.
\newblock Qwen3 technical report.
\newblock \emph{arXiv preprint arXiv:2505.09388}, 2025{\natexlab{b}}.
\bibitem[Touvron et~al.(2023)Touvron, Lavril, Izacard, Martinet, Lachaux,
Lacroix, Rozi{\`e}re, Goyal, Hambro, Azhar, Rodriguez, Joulin, Grave, and
Lample]{touvron-etal:2023llama}
Hugo Touvron, Thibaut Lavril, Gautier Izacard, Xavier Martinet, Marie-Anne
Lachaux, Timoth{\'e}e Lacroix, Baptiste Rozi{\`e}re, Naman Goyal, Eric
Hambro, Faisal Azhar, Aurelien Rodriguez, Armand Joulin, Edouard Grave, and
Guillaume Lample.
\newblock Llama: Open and efficient foundation language models.
\newblock \emph{arXiv preprint arXiv:2302.13971}, 2023.
\bibitem[Uesato et~al.(2022)Uesato, Kushman, Kumar, Song, Siegel, Wang,
Creswell, Irving, and Higgins]{uesato-etal:2022solving}
Jonathan Uesato, Nate Kushman, Ramana Kumar, Francis Song, Noah Siegel, Lisa
Wang, Antonia Creswell, Geoffrey Irving, and Irina Higgins.
\newblock Solving math word problems with process-and outcome-based feedback.
\newblock \emph{arXiv preprint arXiv:2211.14275}, 2022.
\bibitem[van~den Oord et~al.(2018)van~den Oord, Li, and
Vinyals]{OordEtAl2018CPC}
A{"a}ron van~den Oord, Yazhe Li, and Oriol Vinyals.
\newblock Representation learning with contrastive predictive coding.
\newblock \emph{arXiv preprint arXiv:1807.03748}, 2018.
\newblock URL \url{https://arxiv.org/abs/1807.03748}.
\bibitem[Vaswani et~al.(2017{\natexlab{a}})Vaswani, Shazeer, Parmar, Uszkoreit,
Jones, Gomez, Kaiser, and Polosukhin]{VaswaniEtAl2017Transformer}
Ashish Vaswani, Noam Shazeer, Niki Parmar, Jakob Uszkoreit, Llion Jones,
Aidan~N. Gomez, Lukasz Kaiser, and Illia Polosukhin.
\newblock Attention is all you need.
\newblock In \emph{Advances in Neural Information Processing Systems 30}, pages
5998--6008, 2017{\natexlab{a}}.
\newblock URL
\url{https://proceedings.neurips.cc/paper/2017/hash/3f5ee243547dee91fbd053c1c4a845aa-Abstract.html}.
\bibitem[Vaswani et~al.(2017{\natexlab{b}})Vaswani, Shazeer, Parmar, Uszkoreit,
Jones, Gomez, Kaiser, and Polosukhin]{vaswani-etal:2017attention}
Ashish Vaswani, Noam Shazeer, Niki Parmar, Jakob Uszkoreit, Llion Jones,
Aidan~N Gomez, {\L}ukasz Kaiser, and Illia Polosukhin.
\newblock Attention is all you need.
\newblock \emph{Advances in neural information processing systems}, 30,
2017{\natexlab{b}}.
\bibitem[Von~Oswald et~al.(2023)Von~Oswald, Niklasson, Randazzo, Sacramento,
Mordvintsev, Zhmoginov, and Vladymyrov]{von-etal:2023transformers}
Johannes Von~Oswald, Eyvind Niklasson, Ettore Randazzo, Jo{\~a}o Sacramento,
Alexander Mordvintsev, Andrey Zhmoginov, and Max Vladymyrov.
\newblock Transformers learn in-context by gradient descent.
\newblock In \emph{Proceedings of International Conference on Machine
Learning}, pages 35151--35174. PMLR, 2023.
\bibitem[Wang et~al.(2022)Wang, Mishra, Alipoormolabashi, Kordi, Mirzaei, Naik,
Ashok, Dhanasekaran, Arunkumar, Stap, Pathak, Karamanolakis, Lai, Purohit,
Mondal, Anderson, Kuznia, Doshi, Pal, Patel, Moradshahi, Parmar, Purohit,
Varshney, Kaza, Verma, Puri, Karia, Doshi, Sampat, Mishra, A, Patro, Dixit,
and Shen]{wang-etal:2022super}
Yizhong Wang, Swaroop Mishra, Pegah Alipoormolabashi, Yeganeh Kordi, Amirreza
Mirzaei, Atharva Naik, Arjun Ashok, Arut~Selvan Dhanasekaran, Anjana
Arunkumar, David Stap, Eshaan Pathak, Giannis Karamanolakis, Haizhi~Gary Lai,
Ishan Purohit, Ishani Mondal, Jacob Anderson, Kirby Kuznia, Krima Doshi,
Kuntal~Kumar Pal, Maitreya Patel, Mehrad Moradshahi, Mihir Parmar, Mirali
Purohit, Neeraj Varshney, Phani~Rohitha Kaza, Pulkit Verma, Ravsehaj~Singh
Puri, Rushang Karia, Savan Doshi, Shailaja~Keyur Sampat, Siddhartha Mishra,
Sujan~Reddy A, Sumanta Patro, Tanay Dixit, and Xudong Shen.
\newblock Super-naturalinstructions: Generalization via declarative
instructions on 1600+ nlp tasks.
\newblock In \emph{Proceedings of the 2022 Conference on Empirical Methods in
Natural Language Processing}, pages 5085--5109, 2022.
\bibitem[Wang et~al.(2023)Wang, Kordi, Mishra, Liu, Smith, Khashabi, and
Hajishirzi]{wang-etal:2023selfinstruct}
Yizhong Wang, Yeganeh Kordi, Swaroop Mishra, Alisa Liu, Noah~A Smith, Daniel
Khashabi, and Hannaneh Hajishirzi.
\newblock Self-instruct: Aligning language models with self-generated
instructions.
\newblock In \emph{Proceedings of the 61st Annual Meeting of the Association
for Computational Linguistics (Volume 1: Long Papers)}, pages 13484--13508,
2023.
\bibitem[Wei et~al.(2022{\natexlab{a}})Wei, Bosma, Zhao, Guu, Yu, Lester, Du,
Dai, and Le]{wei-etal:2022finetuned}
Jason Wei, Maarten Bosma, Vincent Zhao, Kelvin Guu, Adams~Wei Yu, Brian Lester,
Nan Du, Andrew~M Dai, and Quoc~V Le.
\newblock Finetuned language models are zero-shot learners.
\newblock In \emph{Proceedings of International Conference on Learning
Representations}, 2022{\natexlab{a}}.
\bibitem[Wei et~al.(2022{\natexlab{b}})Wei, Tay, Bommasani, Raffel, Zoph,
Borgeaud, Yogatama, Bosma, Zhou, Metzler, Chi, Hashimoto, Vinyals, Liang,
Dean, and Fedus]{wei-etal:2022emergent}
Jason Wei, Yi~Tay, Rishi Bommasani, Colin Raffel, Barret Zoph, Sebastian
Borgeaud, Dani Yogatama, Maarten Bosma, Denny Zhou, Donald Metzler, Ed~H.
Chi, Tatsunori Hashimoto, Oriol Vinyals, Percy Liang, Jeff Dean, and William
Fedus.
\newblock Emergent abilities of large language models.
\newblock \emph{arXiv preprint arXiv:2206.07682}, 2022{\natexlab{b}}.
\bibitem[Xiao and Zhu(2023)]{xiao-and-zhu:2023introduction}
Tong Xiao and Jingbo Zhu.
\newblock Introduction to transformers: an nlp perspective.
\newblock \emph{arXiv preprint arXiv:2311.17633}, 2023.
\bibitem[Xiao et~al.(2023)Xiao, Wu, Guo, Li, Zhang, Qin, and
Liu]{xiao-etal:2023survey}
Yisheng Xiao, Lijun Wu, Junliang Guo, Juntao Li, Min Zhang, Tao Qin, and
Tie-yan Liu.
\newblock A survey on non-autoregressive generation for neural machine
translation and beyond.
\newblock \emph{IEEE Transactions on Pattern Analysis and Machine
Intelligence}, 45\penalty0 (10):\penalty0 11407--11427, 2023.
\bibitem[Xie et~al.(2022)Xie, Raghunathan, Liang, and Ma]{xie-etal:2022an}
Sang~Michael Xie, Aditi Raghunathan, Percy Liang, and Tengyu Ma.
\newblock An explanation of in-context learning as implicit bayesian inference.
\newblock In \emph{Proceedings of International Conference on Learning
Representations}, 2022.
\bibitem[Xu et~al.(2024)Xu, Sun, Zheng, Geng, Zhao, Feng, Tao, Lin, and
Jiang]{xu-etal:2024wizardlm}
Can Xu, Qingfeng Sun, Kai Zheng, Xiubo Geng, Pu~Zhao, Jiazhan Feng, Chongyang
Tao, Qingwei Lin, and Daxin Jiang.
\newblock Wizardlm: Empowering large pre-trained language models to follow
complex instructions.
\newblock In \emph{The Twelfth International Conference on Learning
Representations}, 2024.
\bibitem[Yan et~al.(2024)Yan, Liu, Pan, Liew, qiang liu, and
Feng]{yan-etal:2024perflow}
Hanshu Yan, Xingchao Liu, Jiachun Pan, Jun~Hao Liew, qiang liu, and Jiashi
Feng.
\newblock Pe{RF}low: Piecewise rectified flow as universal plug-and-play
accelerator.
\newblock In \emph{The Thirty-eighth Annual Conference on Neural Information
Processing Systems}, 2024.
\bibitem[Yang et~al.(2025{\natexlab{a}})Yang, Cheng, Yang, Liu, and
Lin]{yang-etal:2025text}
Xiaofeng Yang, Chen Cheng, Xulei Yang, Fayao Liu, and Guosheng Lin.
\newblock Text-to-image rectified flow as plug-and-play priors.
\newblock In \emph{The Thirteenth International Conference on Learning
Representations}, 2025{\natexlab{a}}.
\bibitem[Yang et~al.(2025{\natexlab{b}})Yang, Wang, Wang, Wen, Qi, Xu, and
Zhang]{yang-etal:2025diffusion}
Yicun Yang, Cong Wang, Shaobo Wang, Zichen Wen, Biqing Qi, Hanlin Xu, and
Linfeng Zhang.
\newblock Diffusion llm with native variable generation lengths: Let [eos] lead
the way, 2025{\natexlab{b}}.
\bibitem[Yao et~al.(2023)Yao, Zhao, Yu, Du, Shafran, Narasimhan, and
Cao]{yao-etal:2023react}
Shunyu Yao, Jeffrey Zhao, Dian Yu, Nan Du, Izhak Shafran, Karthik Narasimhan,
and Yuan Cao.
\newblock React: Synergizing reasoning and acting in language models.
\newblock In \emph{International Conference on Learning Representations
(ICLR)}, 2023.
\bibitem[Yao et~al.(2024)Yao, Yu, Zhao, Shafran, Griffiths, Cao, and
Narasimhan]{yao-etal:2024tree}
Shunyu Yao, Dian Yu, Jeffrey Zhao, Izhak Shafran, Tom Griffiths, Yuan Cao, and
Karthik Narasimhan.
\newblock Tree of thoughts: Deliberate problem solving with large language
models.
\newblock \emph{Advances in Neural Information Processing Systems}, 36, 2024.
\bibitem[Yu et~al.(2023)Yu, He, Wu, Dai, and Chen]{yu-etal:2023towards}
Zihan Yu, Liang He, Zhen Wu, Xinyu Dai, and Jiajun Chen.
\newblock Towards better chain-of-thought prompting strategies: A survey.
\newblock \emph{arXiv preprint arXiv:2310.04959}, 2023.
\bibitem[Zhang et~al.(2025{\natexlab{a}})Zhang, Peng, Zhang, Pan, and
Chrysos]{zhang-etal:2025corrective}
Shuibai Zhang, Fred~Zhangzhi Peng, Yiheng Zhang, Jin Pan, and Grigorios~G
Chrysos.
\newblock Corrective diffusion language models.
\newblock \emph{arXiv preprint arXiv:2512.15596}, 2025{\natexlab{a}}.
\bibitem[Zhang et~al.(2025{\natexlab{b}})Zhang, Tan, Nguyen, Dao, Han, He,
Zhang, Mrdovic, and Metaxas]{zhang-etal:2025:flow}
Xinxi Zhang, Shiwei Tan, Quang Nguyen, Quan Dao, Ligong Han, Xiaoxiao He, Tunyu
Zhang, Alen Mrdovic, and Dimitris Metaxas.
\newblock Flow straighter and faster: Efficient one-step generative modeling
via meanflow on rectified trajectories.
\newblock \emph{arXiv preprint arXiv:2511.23342}, 2025{\natexlab{b}}.
\bibitem[Zhang et~al.(2023)Zhang, Yao, Zhang, Tang, Ma, He, Wang, Gerstein,
Wang, Liu, and Zhao]{zhang-etal:2023igniting}
Zhuosheng Zhang, Yao Yao, Aston Zhang, Xiangru Tang, Xinbei Ma, Zhiwei He,
Yiming Wang, Mark Gerstein, Rui Wang, Gongshen Liu, and Hai Zhao.
\newblock Igniting language intelligence: The hitchhiker's guide from
chain-of-thought reasoning to language agents.
\newblock \emph{arXiv preprint arXiv:2311.11797}, 2023.
\bibitem[Zhou et~al.(2023{\natexlab{a}})Zhou, Liu, Xu, Iyer, Sun, Mao, Ma,
Efrat, Yu, Yu, Zhang, Ghosh, Lewis, Zettlemoyer, and
Levy]{zhou-etal:2023lima}
Chunting Zhou, Pengfei Liu, Puxin Xu, Srini Iyer, Jiao Sun, Yuning Mao, Xuezhe
Ma, Avia Efrat, Ping Yu, Lili Yu, Susan Zhang, Gargi Ghosh, Mike Lewis, Luke
Zettlemoyer, and Omer Levy.
\newblock Lima: Less is more for alignment.
\newblock \emph{arXiv preprint arXiv:2305.11206}, 2023{\natexlab{a}}.
\bibitem[Zhou et~al.(2023{\natexlab{b}})Zhou, Muresanu, Han, Paster, Pitis,
Chan, and Ba]{zhou-etal:2023large}
Yongchao Zhou, Andrei~Ioan Muresanu, Ziwen Han, Keiran Paster, Silviu Pitis,
Harris Chan, and Jimmy Ba.
\newblock Large language models are human-level prompt engineers.
\newblock In \emph{The Eleventh International Conference on Learning
Representations}, 2023{\natexlab{b}}.
\bibitem[Zhu et~al.(2020)Zhu, Cheng, Gan, Sun, Goldstein, and
Liu]{zhu-etal:2020freelb}
Chen Zhu, Yu~Cheng, Zhe Gan, Siqi Sun, Tom Goldstein, and Jingjing Liu.
\newblock Freelb: Enhanced adversarial training for natural language
understanding.
\newblock In \emph{International Conference on Learning Representations}, 2020.
\end{thebibliography}
No preview for this file type
......@@ -239,14 +239,31 @@
\begin{tabular}{rl}
% \textbf{教师:} & \teacher \\
% \textbf{学期:} & \semester \\
\textbf{单位:} & \school \\
\textbf{日期:} & \lessondate \\
% \textbf{单位:} & \school \\
% \textbf{日期:} & \lessondate \\
\end{tabular}
\vspace*{1.2cm}
\end{titlepage}
\frontmatter
\pagestyle{plain}
\chapter*{前言}
\addcontentsline{toc}{chapter}{前言}
机器学习是人工智能领域最重要的基础技术之一,并已广泛应用于自然语言处理、计算机视觉、智能决策等诸多领域。近年来,以大语言模型为代表的生成式人工智能技术快速发展,机器学习的研究对象、方法体系和实现路径也随之发生了显著变化。机器学习已经不再局限于传统的有监督学习以及面向单一任务的模型训练,而是越来越关注大规模数据和计算资源条件下的复杂模型学习,以及模型在开放环境中的生成、推理、决策和交互能力。
本书可以看作是在生成式人工智能时代对机器学习方法的一次重新梳理和介绍。与传统机器学习教材相比,本书不再以经典算法的系统介绍为主要目标,而是更加关注近年来生成式人工智能发展过程中形成的一系列重要方法和技术,包括大模型的训练与微调、深度强化学习、连续时间系统建模、多模态学习等。我们希望通过这些内容,使读者在掌握机器学习基本思想的基础上,进一步理解当前人工智能技术发展的主要方法体系及其背后的基本原理。
本书的部分内容是在我们团队近年来研究和教学工作的基础上整理和发展而来的。在写作过程中,我们参考并吸收了团队此前发表的论文、书籍和技术资料,包括 \href{https://arxiv.org/abs/2501.09223}{\textit{Foundations of Large Language Models}}\href{https://www.techrxiv.org/doi/full/10.36227/techrxiv.177160477.75893679/v1}{\textit{Ordinary Differential Equations in Vision and Language}} 等。在此基础上,本书又根据整体知识体系进行了重新组织,并补充了大量新的内容,希望能够较为完整地反映生成式人工智能背景下机器学习方法的发展脉络。
本书面向对生成式人工智能及现代机器学习方法感兴趣的高年级本科生、研究生、研究人员以及工程技术人员。由于相关技术仍在快速发展,本书难免存在不足之处,也诚挚欢迎读者提出宝贵的意见和建议。
\vspace{1em}
\begin{flushright}
肖桐\quad 王成龙
\end{flushright}
\cleardoublepage
\tableofcontents
\cleardoublepage
......
\chapter{机器学习数学基础与表示学习}
\chapterinfo{李天园、杨曦涵、肖桐}
\chapterinfo[本章内容主要参考 \href{https://niutrans.github.io/NLPBook/}{\textit{NLPBook}} 第一、二章,部分内容翻译、整理自该书英文版]{李天园、杨曦涵、肖桐}
\providecommand{\sourcefigureplaceholder}[2]{%
\begin{figure}[htbp]
......
......@@ -2,7 +2,8 @@
% 第三章正文入口。
\chapter{大模型预训练与微调方法}
\chapterinfo[\href{https://arxiv.org/abs/2501.09223}{\textit{Foundations of Large Language Models}}]{吴钰璋、王骏鑫、王成龙}
% \chapterinfo[\href{https://arxiv.org/abs/2501.09223}{\textit{Foundations of Large Language Models}}]{吴钰璋、王骏鑫、王成龙}
\chapterinfo[本章内容主要参考 \href{https://niutrans.github.io/NLPBook/}{\textit{NLPBook}} 第七、八、九、十章,部分内容翻译、整理自该书英文版]{吴钰璋、王骏鑫、王成龙}
\input{section/3.0-introduction}
\input{section/3.1-pretraining}
......
......@@ -2,7 +2,8 @@
% 本文件由 section4/aml_notes_final.tex 抽取正文生成,保留章节目录独立性。
\chapter{深度学习中的连续动力学机制}
\chapterinfo[\href{https://www.techrxiv.org/doi/full/10.36227/techrxiv.177160477.75893679/v1}{\textit{Ordinary Differential Equations in Vision and Language}}]{张俊翔、叶凯阳、肖桐}
% \chapterinfo[\href{https://www.techrxiv.org/doi/full/10.36227/techrxiv.177160477.75893679/v1}{\textit{Ordinary Differential Equations in Vision and Language}}]{张俊翔、叶凯阳、肖桐}
\chapterinfo[本章内容主要参考 \href{https://www.techrxiv.org/doi/full/10.36227/techrxiv.177160477.75893679/v1}{\textit{Ordinary Differential Equations in Vision and Language}} ,部分内容翻译、整理自该书英文版]{张俊翔、叶凯阳、肖桐}
\begin{learninggoals}
\begin{itemize}
......
Markdown 格式
0%
您添加了 0 到此讨论。请谨慎行事。
请先完成此评论的编辑!
注册 或者 后发表评论