%% refs.bib — Audio-JEPA Representations for Speech Deepfake Detection

%% ---------- phương pháp của ta ----------
@inproceedings{tuncay2025audiojepa,
  author    = {Tuncay, Ludovic and Labb{\'e}, Etienne and Benetos, Emmanouil and Pellegrini, Thomas},
  title     = {Audio-{JEPA}: Joint-Embedding Predictive Architecture for Audio Representation Learning},
  booktitle = {Proc. IEEE International Conference on Multimedia and Expo (ICME) Workshops},
  year      = {2025},
  note      = {arXiv:2507.02915}
}
@inproceedings{assran2023ijepa,
  author    = {Assran, Mahmoud and Duval, Quentin and Misra, Ishan and Bojanowski, Piotr and Vincent, Pascal and Rabbat, Michael and LeCun, Yann and Ballas, Nicolas},
  title     = {Self-Supervised Learning from Images with a Joint-Embedding Predictive Architecture},
  booktitle = {Proc. IEEE/CVF CVPR},
  pages     = {15619--15629},
  year      = {2023}
}
@inproceedings{huang2022audiomae,
  author    = {Huang, Po-Yao and Xu, Hu and Li, Juncheng and Baevski, Alexei and Auli, Michael and Galuba, Wojciech and Metze, Florian and Feichtenhofer, Christoph},
  title     = {Masked Autoencoders that Listen},
  booktitle = {Advances in Neural Information Processing Systems (NeurIPS)},
  volume    = {35},
  year      = {2022}
}
@inproceedings{dosovitskiy2021vit,
  author    = {Dosovitskiy, Alexey and Beyer, Lucas and Kolesnikov, Alexander and Weissenborn, Dirk and Zhai, Xiaohua and Unterthiner, Thomas and Dehghani, Mostafa and Minderer, Matthias and Heigold, Georg and Gelly, Sylvain and Uszkoreit, Jakob and Houlsby, Neil},
  title     = {An Image is Worth 16x16 Words: Transformers for Image Recognition at Scale},
  booktitle = {Proc. ICLR},
  year      = {2021}
}
@inproceedings{gemmeke2017audioset,
  author    = {Gemmeke, Jort F. and Ellis, Daniel P. W. and Freedman, Dylan and Jansen, Aren and Lawrence, Wade and Moore, R. Channing and Plakal, Manoj and Ritter, Marvin},
  title     = {Audio Set: An Ontology and Human-Labeled Dataset for Audio Events},
  booktitle = {Proc. IEEE ICASSP},
  pages     = {776--780},
  year      = {2017}
}
@inproceedings{okabe2018asp,
  author    = {Okabe, Koji and Koshinaka, Takafumi and Shinoda, Koichi},
  title     = {Attentive Statistics Pooling for Deep Speaker Embedding},
  booktitle = {Proc. Interspeech},
  pages     = {2252--2256},
  year      = {2018}
}
@inproceedings{park2019specaugment,
  author    = {Park, Daniel S. and Chan, William and Zhang, Yu and Chiu, Chung-Cheng and Zoph, Barret and Cubuk, Ekin D. and Le, Quoc V.},
  title     = {{SpecAugment}: A Simple Data Augmentation Method for Automatic Speech Recognition},
  booktitle = {Proc. Interspeech},
  pages     = {2613--2617},
  year      = {2019}
}
@inproceedings{loshchilov2019adamw,
  author    = {Loshchilov, Ilya and Hutter, Frank},
  title     = {Decoupled Weight Decay Regularization},
  booktitle = {Proc. ICLR},
  year      = {2019}
}

%% ---------- benchmark và bộ công cụ ----------
@misc{lecun2022path,
  author       = {LeCun, Yann},
  title        = {A Path Towards Autonomous Machine Intelligence},
  year         = {2022},
  howpublished = {OpenReview},
  note         = {Version 0.9.2, 27 June 2022},
  url          = {https://openreview.net/forum?id=BZ5a1r-kVsf}
}
@misc{fei2023ajepa,
  author       = {Fei, Zhengcong and Fan, Mingyuan and Huang, Junshi},
  title        = {{A-JEPA}: Joint-Embedding Predictive Architecture Can Listen},
  year         = {2023},
  howpublished = {arXiv:2311.15830}
}
@misc{dowerah2025arena,
  author    = {Dowerah, Sandipana and Kulkarni, Atharva and Kulkarni, Ajinkya and Tran, Hoan My and Kalda, Joonas and Fedorchenko, Artem and Fauve, Benoit and Lolive, Damien and Alum{\"a}e, Tanel and Magimai Doss, Matthew},
  title     = {Speech {DF} Arena: A Leaderboard for Speech Deepfake Detection Models},
  year      = {2025},
  howpublished = {arXiv:2509.02859v1}
}
@misc{arenaleaderboard2026,
  author    = {{Speech Arena}},
  title     = {Speech {DF} Arena Leaderboard},
  year      = {2026},
  howpublished = {\url{https://huggingface.co/spaces/Speech-Arena-2025/Speech-DF-Arena}},
  note      = {Accessed 9 September 2026}
}
@inproceedings{todisco2019asvspoof,
  author    = {Todisco, Massimiliano and Wang, Xin and Vestman, Ville and Sahidullah, Md and Delgado, H{\'e}ctor and Nautsch, Andreas and Yamagishi, Junichi and Evans, Nicholas and Kinnunen, Tomi and Lee, Kong Aik},
  title     = {{ASVspoof} 2019: Future Horizons in Spoofed and Fake Audio Detection},
  booktitle = {Proc. Interspeech},
  pages     = {1008--1012},
  year      = {2019}
}
@article{wang2020asvspoof,
  author    = {Wang, Xin and Yamagishi, Junichi and Todisco, Massimiliano and Delgado, H{\'e}ctor and Nautsch, Andreas and Evans, Nicholas and Sahidullah, Md and Vestman, Ville and Kinnunen, Tomi and Lee, Kong Aik and others},
  title     = {{ASVspoof} 2019: A Large-Scale Public Database of Synthesized, Converted and Replayed Speech},
  journal   = {Computer Speech \& Language},
  volume    = {64},
  pages     = {101114},
  year      = {2020}
}
@inproceedings{muller2022itw,
  author    = {M{\"u}ller, Nicolas M. and Czempin, Pavel and Dieckmann, Franziska and Froghyar, Adam and B{\"o}ttinger, Konstantin},
  title     = {Does Audio Deepfake Detection Generalize?},
  booktitle = {Proc. Interspeech},
  pages     = {2783--2787},
  year      = {2022}
}
@inproceedings{du2024dfadd,
  author    = {Du, Jiawei and Lin, I-Ming and Chiu, I-Hsiang and Chen, Xuanjun and Wu, Haibin and Ren, Wenze and Tsao, Yu and Lee, Hung-yi and Jang, Jyh-Shing Roger},
  title     = {{DFADD}: The Diffusion and Flow-Matching Based Audio Deepfake Dataset},
  booktitle = {Proc. IEEE Spoken Language Technology Workshop (SLT)},
  pages     = {921--928},
  doi       = {10.1109/SLT61566.2024.10832250},
  year      = {2024},
  note      = {arXiv:2409.08731}
}
@inproceedings{sun2023librisevoc,
  author    = {Sun, Chengzhe and Jia, Shan and Hou, Shuwei and Lyu, Siwei},
  title     = {{AI}-Synthesized Voice Detection Using Neural Vocoder Artifacts},
  booktitle = {Proc. IEEE/CVF CVPR Workshops},
  pages     = {904--912},
  year      = {2023},
  note      = {arXiv:2304.13085}
}
@inproceedings{yi2022add,
  author    = {Yi, Jiangyan and Fu, Ruibo and Tao, Jianhua and Nie, Shuai and Ma, Haoxin and Wang, Chenglong and Wang, Tao and Tian, Zhengkun and Bai, Ye and Fan, Cunhang and others},
  title     = {{ADD} 2022: The First Audio Deep Synthesis Detection Challenge},
  booktitle = {Proc. IEEE ICASSP},
  pages     = {9216--9220},
  year      = {2022}
}
@misc{veaux2017vctk,
  author    = {Yamagishi, Junichi and Veaux, Christophe and MacDonald, Kirsten},
  title     = {{CSTR VCTK} Corpus: English Multi-speaker Corpus for {CSTR} Voice Cloning Toolkit (version 0.92)},
  howpublished = {University of Edinburgh, The Centre for Speech Technology Research (CSTR), sound dataset},
  doi       = {10.7488/ds/2645},
  year      = {2019}
}
@inproceedings{zen2019libritts,
  author    = {Zen, Heiga and Dang, Viet and Clark, Rob and Zhang, Yu and Weiss, Ron J. and Jia, Ye and Chen, Zhifeng and Wu, Yonghui},
  title     = {{LibriTTS}: A Corpus Derived from {LibriSpeech} for Text-to-Speech},
  booktitle = {Proc. Interspeech},
  pages     = {1526--1530},
  year      = {2019}
}

%% ---------- baseline và hệ đối chiếu ----------
@inproceedings{jung2022aasist,
  author    = {Jung, Jee-weon and Heo, Hee-Soo and Tak, Hemlata and Shim, Hye-jin and Chung, Joon Son and Lee, Bong-Jin and Yu, Ha-Jin and Evans, Nicholas},
  title     = {{AASIST}: Audio Anti-Spoofing Using Integrated Spectro-Temporal Graph Attention Networks},
  booktitle = {Proc. IEEE ICASSP},
  pages     = {6367--6371},
  year      = {2022}
}
@inproceedings{tak2021rawnet2,
  author    = {Tak, Hemlata and Patino, Jose and Todisco, Massimiliano and Nautsch, Andreas and Evans, Nicholas and Larcher, Anthony},
  title     = {End-to-End Anti-Spoofing with {RawNet2}},
  booktitle = {Proc. IEEE ICASSP},
  pages     = {6369--6373},
  year      = {2021}
}
@inproceedings{tak2021rawgatst,
  author    = {Tak, Hemlata and Jung, Jee-weon and Patino, Jose and Kamble, Madhu and Todisco, Massimiliano and Evans, Nicholas},
  title     = {End-to-End Spectro-Temporal Graph Attention Networks for Speaker Verification Anti-Spoofing and Speech Deepfake Detection},
  booktitle = {Proc. ASVspoof 2021 Workshop},
  pages     = {1--8},
  year      = {2021}
}
@inproceedings{tak2022w2v2aasist,
  author    = {Tak, Hemlata and Todisco, Massimiliano and Wang, Xin and Jung, Jee-weon and Yamagishi, Junichi and Evans, Nicholas},
  title     = {Automatic Speaker Verification Spoofing and Deepfake Detection Using wav2vec 2.0 and Data Augmentation},
  booktitle = {Proc. The Speaker and Language Recognition Workshop (Odyssey)},
  pages     = {112--119},
  year      = {2022}
}
@inproceedings{zhang2024xlsrsls,
  author    = {Zhang, Qishan and Wen, Shuangbing and Hu, Tao},
  title     = {Audio Deepfake Detection with Self-Supervised {XLS-R} and {SLS} Classifier},
  booktitle = {Proc. ACM International Conference on Multimedia (MM)},
  pages     = {6765--6773},
  doi       = {10.1145/3664647.3681345},
  year      = {2024}
}
@inproceedings{truong2024tcm,
  author    = {Truong, Duc-Tuan and Tao, Ruijie and Nguyen, Tuan and Luong, Hieu-Thi and Lee, Kong Aik and Chng, Eng Siong},
  title     = {Temporal-Channel Modeling in Multi-head Self-Attention for Synthetic Speech Detection},
  booktitle = {Proc. Interspeech},
  year      = {2024}
}
@article{xiao2025xlsrmamba,
  author    = {Xiao, Yang and Das, Rohan Kumar},
  title     = {{XLSR-Mamba}: A Dual-Column Bidirectional State Space Model for Spoofing Attack Detection},
  journal   = {IEEE Signal Processing Letters},
  volume    = {32},
  pages     = {1276--1280},
  doi       = {10.1109/LSP.2025.3547861},
  year      = {2025}
}
@article{liu2025nes2net,
  author    = {Liu, Tianchi and Truong, Duc-Tuan and Das, Rohan Kumar and Lee, Kong Aik and Li, Haizhou},
  title     = {{Nes2Net}: A Lightweight Nested Architecture for Foundation Model Driven Speech Anti-Spoofing},
  year      = {2025},
  journal   = {IEEE Transactions on Information Forensics and Security},
  volume    = {20},
  pages     = {12005--12018},
  doi       = {10.1109/TIFS.2025.3626963}
}
@inproceedings{kawa2023whisper,
  author    = {Kawa, Piotr and Plata, Marcin and Czuba, Micha{\l} and Szyma{\'n}ski, Piotr and Syga, Piotr},
  title     = {Improved {DeepFake} Detection Using {Whisper} Features},
  booktitle = {Proc. Interspeech},
  pages     = {4009--4013},
  year      = {2023}
}
@inproceedings{desplanques2020ecapa,
  author    = {Desplanques, Brecht and Thienpondt, Jenthe and Demuynck, Kris},
  title     = {{ECAPA-TDNN}: Emphasized Channel Attention, Propagation and Aggregation in {TDNN} Based Speaker Verification},
  booktitle = {Proc. Interspeech},
  pages     = {3830--3834},
  year      = {2020}
}
@misc{lin2026teffic,
  author    = {Lin, Wan and Wang, Li and Wang, Jindong and Feng, Kunyu and Wu, Zhizheng},
  title     = {{Teffic-Audio}: Tell Fact from Fiction},
  year      = {2026},
  howpublished = {arXiv:2607.28351v2}
}

%% ---------- front-end SSL tiếng nói ----------
@inproceedings{baevski2020wav2vec2,
  author    = {Baevski, Alexei and Zhou, Yuhao and Mohamed, Abdelrahman and Auli, Michael},
  title     = {wav2vec 2.0: A Framework for Self-Supervised Learning of Speech Representations},
  booktitle = {Advances in Neural Information Processing Systems (NeurIPS)},
  volume    = {33},
  year      = {2020}
}
@inproceedings{babu2022xlsr,
  author    = {Babu, Arun and Wang, Changhan and Tjandra, Andros and Lakhotia, Kushal and Xu, Qiantong and Goyal, Naman and Singh, Kritika and von Platen, Patrick and Saraf, Yatharth and Pino, Juan and Baevski, Alexei and Conneau, Alexis and Auli, Michael},
  title     = {{XLS-R}: Self-Supervised Cross-Lingual Speech Representation Learning at Scale},
  booktitle = {Proc. Interspeech},
  pages     = {2278--2282},
  year      = {2022}
}
@article{chen2022wavlm,
  author    = {Chen, Sanyuan and Wang, Chengyi and Chen, Zhengyang and Wu, Yu and Liu, Shujie and Chen, Zhuo and Li, Jinyu and Kanda, Naoyuki and Yoshioka, Takuya and Xiao, Xiong and others},
  title     = {{WavLM}: Large-Scale Self-Supervised Pre-Training for Full Stack Speech Processing},
  journal   = {IEEE Journal of Selected Topics in Signal Processing},
  volume    = {16},
  number    = {6},
  pages     = {1505--1518},
  year      = {2022}
}
@article{hsu2021hubert,
  author    = {Hsu, Wei-Ning and Bolte, Benjamin and Tsai, Yao-Hung Hubert and Lakhotia, Kushal and Salakhutdinov, Ruslan and Mohamed, Abdelrahman},
  title     = {{HuBERT}: Self-Supervised Speech Representation Learning by Masked Prediction of Hidden Units},
  journal   = {IEEE/ACM Transactions on Audio, Speech, and Language Processing},
  volume    = {29},
  pages     = {3451--3460},
  year      = {2021}
}
@inproceedings{wang2022ssl_frontends,
  author    = {Wang, Xin and Yamagishi, Junichi},
  title     = {Investigating Self-Supervised Front Ends for Speech Spoofing Countermeasures},
  booktitle = {Proc. The Speaker and Language Recognition Workshop (Odyssey)},
  pages     = {100--106},
  year      = {2022}
}
@misc{yang2024features,
  author    = {Yang, Yujie and others},
  title     = {A Robust Audio Deepfake Detection System via Multi-View Feature},
  year      = {2024},
  howpublished = {arXiv:2403.01960}
}
@inproceedings{bardes2024vjepa,
  author    = {Bardes, Adrien and Garrido, Quentin and Ponce, Jean and Chen, Xinlei and Rabbat, Michael and LeCun, Yann and Assran, Mahmoud and Ballas, Nicolas},
  title     = {Revisiting Feature Prediction for Learning Visual Representations from Video},
  booktitle = {Transactions on Machine Learning Research},
  year      = {2024}
}
