\bibitem{simclr} Chen T, et al. A Simple Framework for Contrastive Learning of Visual Representations (SimCLR). ICML, 2020.
\bibitem{moco} He K, et al. Momentum Contrast for Unsupervised Visual Representation Learning (MoCo). CVPR, 2020.
\bibitem{dino} Caron M, et al. Emerging Properties in Self-Supervised Vision Transformers (DINO). ICCV, 2021.
\bibitem{cpc} van den Oord A, Li Y, Vinyals O. Representation Learning with Contrastive Predictive Coding (CPC). arXiv:1807.03748, 2018.
\bibitem{wav2vec2} Baevski A, Zhou H, Mohamed A, Auli M. wav2vec 2.0: A Framework for Self-Supervised Learning of Speech Representations. NeurIPS, 2020.
\bibitem{hubert} Hsu W N, et al. HuBERT: Self-Supervised Speech Representation Learning by Masked Prediction of Hidden Units. IEEE/ACM Transactions on Audio, Speech, and Language Processing, 2021.
\bibitem{data2vec} Baevski A, et al. data2vec: A General Framework for Self-Supervised Learning in Speech, Vision and Language. ICML, 2022.
\bibitem{mae} He K, et al. Masked Autoencoders Are Scalable Vision Learners (MAE). CVPR, 2022.
\bibitem{beit} Bao H, Dong L, Piao S, Wei F. BEiT: BERT Pre-Training of Image Transformers. ICLR, 2022.
\bibitem{clip} Radford A, et al. Learning Transferable Visual Models From Natural Language Supervision (CLIP). ICML, 2021.
\bibitem{align} Jia C, et al. Scaling Up Visual and Vision-Language Representation Learning With Noisy Text Supervision (ALIGN). ICML, 2021.
\bibitem{siglip} Zhai X, et al. Sigmoid Loss for Language Image Pre-Training (SigLIP). ICCV, 2023.
\bibitem{waco} Ouyang S, Ye R, Li L. WACO: Word-Aligned Contrastive Learning for Speech Translation. ACL, 2023.
\bibitem{clap} Elizalde B, Deshmukh S, Al Ismail M, Wang H. CLAP: Learning Audio Concepts From Natural Language Supervision. ICASSP, 2023.
\bibitem{flamingo} Alayrac J B, et al. Flamingo: a Visual Language Model for Few-Shot Learning. NeurIPS, 2022.
\bibitem{blip2} Li J, et al. BLIP-2: Bootstrapping Language-Image Pre-training with Frozen Image Encoders and Large Language Models. ICML, 2023.
\bibitem{llava} Liu H, et al. Visual Instruction Tuning (LLaVA). NeurIPS, 2023.
\bibitem{instructblip} Dai W, et al. InstructBLIP: Towards General-purpose Vision-Language Models with Instruction Tuning. NeurIPS, 2023.
\bibitem{qwenvl} Bai J, et al. Qwen-VL: A Versatile Vision-Language Model for Understanding, Localization, Text Reading, and Beyond. arXiv:2308.12966, 2023.
\bibitem{whisper} Radford A, et al. Robust Speech Recognition via Large-Scale Weak Supervision (Whisper). ICML, 2023.
\bibitem{speechgpt} Zhang D, et al. SpeechGPT: Empowering Large Language Models with Intrinsic Cross-Modal Conversational Abilities. Findings of EMNLP, 2023.
\bibitem{qwenaudio} Chu Y, et al. Qwen-Audio: Advancing Universal Audio Understanding via Unified Large-Scale Audio-Language Models. arXiv:2311.07919, 2023.
\bibitem{salmonn} Tang C, et al. SALMONN: Towards Generic Hearing Abilities for Large Language Models. ICLR, 2024.
\bibitem{emu} Sun Q, et al. Emu: Generative Pretraining in Multimodality. ICLR, 2024.
\bibitem{gan} Goodfellow I, et al. Generative Adversarial Nets. NeurIPS, 2014.
\bibitem{vae} Kingma D P, Welling M. Auto-Encoding Variational Bayes (VAE). ICLR, 2014.
...
...
@@ -367,8 +377,19 @@
\bibitem{p2p} Hertz A, et al. Prompt-to-Prompt Image Editing with Cross Attention Control. ICLR, 2023.
\bibitem{ip2p} Brooks T, Holynski A, Efros A A. InstructPix2Pix: Learning to Follow Image Editing Instructions. CVPR, 2023.
\bibitem{controlnet} Zhang L, Rao A, Agrawala M. Adding Conditional Control to Text-to-Image Diffusion Models (ControlNet). ICCV, 2023.
\bibitem{tacotron2} Shen J, et al. Natural TTS Synthesis by Conditioning WaveNet on Mel Spectrogram Predictions (Tacotron 2). ICASSP, 2018.
\bibitem{fastspeech2} Ren Y, et al. FastSpeech 2: Fast and High-Quality End-to-End Text to Speech. ICLR, 2021.
\bibitem{vits} Kim J, Kong J, Son J. Conditional Variational Autoencoder with Adversarial Learning for End-to-End Text-to-Speech (VITS). ICML, 2021.
\bibitem{valle} Wang C, et al. Neural Codec Language Models are Zero-Shot Text to Speech Synthesizers (VALL-E). arXiv:2301.02111, 2023.
\bibitem{nist-synthetic} National Institute of Standards and Technology. Reducing Risks Posed by Synthetic Content. NIST AI 100-4, 2024.
\bibitem{dreambooth} Ruiz N, et al. DreamBooth: Fine Tuning Text-to-Image Diffusion Models for Subject-Driven Generation. CVPR, 2023.