@article{bb384800,
AUTHOR = "Lee, J. and Shin, Y. and Chang, J.H.",
TITLE = "Differentiable Duration Refinement Using Internal Division for
Non-Autoregressive Text-to-Speech",
JOURNAL = SPLetters,
VOLUME = "31",
YEAR = "2024",
PAGES = "3154-3158",
BIBSOURCE = "http://www.visionbib.com/bibliography/other1024ss1.html#TT378881"}
@article{bb384801,
AUTHOR = "Xu, X. and Ma, Z.Y. and Wu, M.Y. and Yu, K.",
TITLE = "Towards Weakly Supervised Text-to-Audio Grounding",
JOURNAL = MultMed,
VOLUME = "26",
YEAR = "2024",
PAGES = "11126-11138",
BIBSOURCE = "http://www.visionbib.com/bibliography/other1024ss1.html#TT378882"}
@article{bb384802,
AUTHOR = "Kim, M. and Jeong, M. and Lee, J.Y. and Kim, N.S.",
TITLE = "SegINR: Segment-Wise Implicit Neural Representation for Sequence
Alignment in Neural Text-to-Speech",
JOURNAL = SPLetters,
VOLUME = "32",
YEAR = "2025",
PAGES = "646-650",
BIBSOURCE = "http://www.visionbib.com/bibliography/other1024ss1.html#TT378883"}
@article{bb384803,
AUTHOR = "Zheng, J.J. and Zhou, J. and Zheng, W.M. and Tao, L. and Kwan, H.K.",
TITLE = "Controllable Multi-Speaker Emotional Speech Synthesis With an Emotion
Representation of High Generalization Capability",
JOURNAL = AffCom,
VOLUME = "16",
YEAR = "2025",
NUMBER = "1",
MONTH = "January",
PAGES = "68-82",
BIBSOURCE = "http://www.visionbib.com/bibliography/other1024ss1.html#TT378884"}
@article{bb384804,
AUTHOR = "Chen, K. and Huang, Z.H. and He, L. and Yan, Y.H.",
TITLE = "UnitDiff: A Unit-Diffusion Model for Code-Switching Speech Synthesis",
JOURNAL = SPLetters,
VOLUME = "32",
YEAR = "2025",
PAGES = "1051-1055",
BIBSOURCE = "http://www.visionbib.com/bibliography/other1024ss1.html#TT378885"}
@article{bb384805,
AUTHOR = "Chang, Y. and Ko, Y.J.",
TITLE = "Soft engagement with pseudo initiatives for multi-party dialogue
generation",
JOURNAL = PRL,
VOLUME = "191",
YEAR = "2025",
PAGES = "103-109",
BIBSOURCE = "http://www.visionbib.com/bibliography/other1024ss1.html#TT378886"}
@article{bb384806,
AUTHOR = "He, Y.L. and Wang, H.X. and Qiu, Y.Q. and Cao, H.",
TITLE = "ASSMark: Dual Defense Against Speech Synthesis Attack via Adversarial
Robust Watermarking",
JOURNAL = SPLetters,
VOLUME = "32",
YEAR = "2025",
PAGES = "1870-1874",
BIBSOURCE = "http://www.visionbib.com/bibliography/other1024ss1.html#TT378887"}
@article{bb384807,
AUTHOR = "Wang, R. and Chen, L.P. and Lee, K.A. and Ling, Z.H.",
TITLE = "Asynchronous Voice Anonymization by Learning From Speaker-Adversarial
Speech",
JOURNAL = SPLetters,
VOLUME = "32",
YEAR = "2025",
PAGES = "1905-1909",
BIBSOURCE = "http://www.visionbib.com/bibliography/other1024ss1.html#TT378888"}
@article{bb384808,
AUTHOR = "Feng, Y. and Zhang, X.B. and Feng, F.Y. and Zhang, G.L. and Xu, L.T.",
TITLE = "Robust and Imperceptible Watermarking Framework for Generative Audio
Models",
JOURNAL = SPLetters,
VOLUME = "32",
YEAR = "2025",
PAGES = "3196-3200",
BIBSOURCE = "http://www.visionbib.com/bibliography/other1024ss1.html#TT378889"}
@article{bb384809,
AUTHOR = "Lee, J. and Song, N.S. and Chang, J.H.",
TITLE = "Vector Field Decomposition-Based Flow Matching for Zero-Shot
Cross-Lingual Text-to-Speech",
JOURNAL = SPLetters,
VOLUME = "32",
YEAR = "2025",
PAGES = "3560-3564",
BIBSOURCE = "http://www.visionbib.com/bibliography/other1024ss1.html#TT378890"}
@article{bb384810,
AUTHOR = "Wang, H. and Yang, Y.F. and Liu, S. and Li, J. and Meng, L. and Liu, Y.Q. and Zhou, J.M. and Sun, H.Q. and Lu, Y. and Qin, Y.",
TITLE = "StreamMel: Real-Time Zero-Shot Text-to-Speech Via Interleaved
Continuous Autoregressive Modeling",
JOURNAL = SPLetters,
VOLUME = "32",
YEAR = "2025",
PAGES = "3530-3534",
BIBSOURCE = "http://www.visionbib.com/bibliography/other1024ss1.html#TT378891"}
@article{bb384811,
AUTHOR = "Li, L. and Cong, G.X. and Qi, Y.K. and Zha, Z.J. and Wu, Q. and Sheng, Q.Z. and Huang, Q.M. and Yang, M.H.",
TITLE = "Dubbing Movies via Hierarchical Phoneme Modeling and Acoustic
Diffusion Denoising",
JOURNAL = PAMI,
VOLUME = "47",
YEAR = "2025",
NUMBER = "11",
MONTH = "November",
PAGES = "10361-10377",
BIBSOURCE = "http://www.visionbib.com/bibliography/other1024ss1.html#TT378892"}
@article{bb384812,
AUTHOR = "Gao, X.X. and Zhang, H. and Chen, N.F.",
TITLE = "Prompt-Unseen-Emotion: Mixed Emotional Speech Synthesis With
Prompt-LLM Contextual Knowledge",
JOURNAL = SPLetters,
VOLUME = "32",
YEAR = "2025",
PAGES = "4259-4263",
BIBSOURCE = "http://www.visionbib.com/bibliography/other1024ss1.html#TT378893"}
@article{bb384813,
AUTHOR = "Lee, K. and Hong, S. and Chun, S.Y.",
TITLE = "Robust watermarks for audio diffusion models by quadrature amplitude
modulation",
JOURNAL = PRL,
VOLUME = "198",
YEAR = "2025",
PAGES = "22-28",
BIBSOURCE = "http://www.visionbib.com/bibliography/other1024ss1.html#TT378894"}
@article{bb384814,
AUTHOR = "Inoue, S. and Zhou, K. and Wang, S. and Li, H.Z.",
TITLE = "Hierarchical Control of Emotion Rendering in Speech Synthesis",
JOURNAL = AffCom,
VOLUME = "16",
YEAR = "2025",
NUMBER = "4",
MONTH = "October",
PAGES = "3316-3328",
BIBSOURCE = "http://www.visionbib.com/bibliography/other1024ss1.html#TT378895"}
@article{bb384815,
AUTHOR = "Cha, H. and Um, S. and Kim, M. and Kim, C. and Lee, S. and Kang, H.G.",
TITLE = "Content-Aware Style Augmentation for Zero-Shot Voice Conversion With
Short Target Speech",
JOURNAL = SPLetters,
VOLUME = "33",
YEAR = "2026",
PAGES = "66-70",
BIBSOURCE = "http://www.visionbib.com/bibliography/other1024ss1.html#TT378896"}
@article{bb384816,
AUTHOR = "Haji Ali, M. and Menapace, W. and Siarohin, A. and Balakrishnan, G. and Ordonez, V.",
TITLE = "Taming Data and Transformers for Audio Generation",
JOURNAL = IJCV,
VOLUME = "134",
YEAR = "2026",
NUMBER = "1",
MONTH = "January",
PAGES = "87",
BIBSOURCE = "http://www.visionbib.com/bibliography/other1024ss1.html#TT378897"}
@inproceedings{bb384817,
AUTHOR = "Liu, J. and Geddes, J. and Guo, Z.Y. and Jiang, H. and Nandwana, M.K.",
TITLE = "Smooth Cache: A Universal Inference Acceleration Technique for
Diffusion Transformers",
BOOKTITLE = LargeVM25,
YEAR = "2025",
PAGES = "3220-3229",
BIBSOURCE = "http://www.visionbib.com/bibliography/other1024ss1.html#TT378898"}
@inproceedings{bb384818,
AUTHOR = "Kushwaha, S.S. and Tian, Y.P.",
TITLE = "VinTAGe: Joint Video and Text Conditioning for Holistic Audio
Generation",
BOOKTITLE = CVPR25,
YEAR = "2025",
PAGES = "13529-13539",
BIBSOURCE = "http://www.visionbib.com/bibliography/other1024ss1.html#TT378899"}
@inproceedings{bb384819,
AUTHOR = "Kim, J.H. and Choi, J. and Kim, J.H. and Jung, C. and Chung, J.S.",
TITLE = "From Faces to Voices: Learning Hierarchical Representations for
High-quality Video-to-Speech",
BOOKTITLE = CVPR25,
YEAR = "2025",
PAGES = "15874-15884",
BIBSOURCE = "http://www.visionbib.com/bibliography/other1024ss1.html#TT378900"}
@inproceedings{bb384820,
AUTHOR = "Cong, G.X. and Pan, J. and Li, L. and Qi, Y.K. and Peng, Y.X. and van den Hengel, A.J. and Yang, J. and Huang, Q.M.",
TITLE = "EmoDubber: Towards High Quality and Emotion Controllable Movie
Dubbing",
BOOKTITLE = CVPR25,
YEAR = "2025",
PAGES = "15863-15873",
BIBSOURCE = "http://www.visionbib.com/bibliography/other1024ss1.html#TT378901"}
@inproceedings{bb384821,
AUTHOR = "Zhang, Z.D. and Li, L. and Yan, C.G. and Liu, C.S. and van den Hengel, A.J. and Qi, Y.K.",
TITLE = "Prosody-Enhanced Acoustic Pre-training and Acoustic-Disentangled
Prosody Adapting for Movie Dubbing",
BOOKTITLE = CVPR25,
YEAR = "2025",
PAGES = "172-182",
BIBSOURCE = "http://www.visionbib.com/bibliography/other1024ss1.html#TT378902"}
@inproceedings{bb384822,
AUTHOR = "Rai, A. and Sridhar, S.",
TITLE = "EgoSonics: Generating Synchronized Audio for Silent Egocentric Videos",
BOOKTITLE = WACV25,
YEAR = "2025",
PAGES = "4935-4946",
BIBSOURCE = "http://www.visionbib.com/bibliography/other1024ss1.html#TT378903"}
@inproceedings{bb384823,
AUTHOR = "Yadav, A.K.S. and Bhagtani, K. and Salvi, D. and Bestagini, P. and Delp, E.J.",
TITLE = "FairSSD: Understanding Bias in Synthetic Speech Detectors",
BOOKTITLE = WMF24,
YEAR = "2024",
PAGES = "4418-4428",
BIBSOURCE = "http://www.visionbib.com/bibliography/other1024ss1.html#TT378904"}
@inproceedings{bb384824,
AUTHOR = "Cuccovillo, L. and Gerhardt, M. and Aichroth, P.",
TITLE = "Audio Transformer for Synthetic Speech Detection via Multi-Formant
Analysis",
BOOKTITLE = WMF24,
YEAR = "2024",
PAGES = "4409-4417",
BIBSOURCE = "http://www.visionbib.com/bibliography/other1024ss1.html#TT378905"}
@inproceedings{bb384825,
AUTHOR = "Cong, G.X. and Li, L. and Qi, Y.K. and Zha, Z.J. and Wu, Q. and Wang, W.Y. and Jiang, B. and Yang, M.H. and Huang, Q.M.",
TITLE = "Learning to Dub Movies via Hierarchical Prosody Models",
BOOKTITLE = CVPR23,
YEAR = "2023",
PAGES = "14687-14697",
BIBSOURCE = "http://www.visionbib.com/bibliography/other1024ss1.html#TT378906"}
@inproceedings{bb384826,
AUTHOR = "Hsu, W.N. and Remez, T. and Shi, B. and Donley, J. and Adi, Y.",
TITLE = "ReVISE: Self-Supervised Speech Resynthesis with Visual Input for
Universal and Generalized Speech Regeneration",
BOOKTITLE = CVPR23,
YEAR = "2023",
PAGES = "18796-18806",
BIBSOURCE = "http://www.visionbib.com/bibliography/other1024ss1.html#TT378907"}
@inproceedings{bb384827,
AUTHOR = "Sun, C.Z. and Jia, S. and Hou, S.W. and Lyu, S.W.",
TITLE = "AI-Synthesized Voice Detection Using Neural Vocoder Artifacts",
BOOKTITLE = WMF23,
YEAR = "2023",
PAGES = "904-912",
BIBSOURCE = "http://www.visionbib.com/bibliography/other1024ss1.html#TT378908"}
@inproceedings{bb384828,
AUTHOR = "Noufi, C. and May, L. and Berger, J.",
TITLE = "The Role of Vocal Persona in Natural and Synthesized Speech",
BOOKTITLE = FG23,
YEAR = "2023",
PAGES = "1-4",
BIBSOURCE = "http://www.visionbib.com/bibliography/other1024ss1.html#TT378909"}
@inproceedings{bb384829,
AUTHOR = "Hwang, I.S. and Lee, S.H. and Lee, S.W.",
TITLE = "StyleVC: Non-Parallel Voice Conversion with Adversarial Style
Generalization",
BOOKTITLE = "ICPR22",
YEAR = "2022",
PAGES = "23-30",
BIBSOURCE = "http://www.visionbib.com/bibliography/other1024ss1.html#TT378910"}
@inproceedings{bb384830,
AUTHOR = "Wang, W.B. and Song, Y. and Jha, S.",
TITLE = "Autolv: Automatic Lecture Video Generator",
BOOKTITLE = ICIP22,
YEAR = "2022",
PAGES = "1086-1090",
BIBSOURCE = "http://www.visionbib.com/bibliography/other1024ss1.html#TT378911"}
@inproceedings{bb384831,
AUTHOR = "Borzi, S. and Giudice, O. and Stanco, F. and Allegra, D.",
TITLE = "Is synthetic voice detection research going into the right direction?",
BOOKTITLE = WMF22,
YEAR = "2022",
PAGES = "71-80",
BIBSOURCE = "http://www.visionbib.com/bibliography/other1024ss1.html#TT378912"}
@inproceedings{bb384832,
AUTHOR = "Hassid, M. and Ramanovich, M.T. and Shillingford, B. and Wang, M. and Jia, Y. and Remez, T.",
TITLE = "More than Words: In-the-Wild Visually-Driven Prosody for
Text-to-Speech",
BOOKTITLE = CVPR22,
YEAR = "2022",
PAGES = "10577-10587",
BIBSOURCE = "http://www.visionbib.com/bibliography/other1024ss1.html#TT378913"}
@inproceedings{bb384833,
AUTHOR = "Kwak, I.Y. and Kwag, S. and Lee, J. and Huh, J.H. and Lee, C.H. and Jeon, Y.B. and Hwang, J.H. and Yoon, J.W.",
TITLE = "ResMax: Detecting Voice Spoofing Attacks with Residual Network and
Max Feature Map",
BOOKTITLE = ICPR21,
YEAR = "2021",
PAGES = "4837-4844",
BIBSOURCE = "http://www.visionbib.com/bibliography/other1024ss1.html#TT378914"}
@inproceedings{bb384834,
AUTHOR = "Wang, D.H. and Wang, R. and Dong, L. and Yan, D. and Ren, Y.M.",
TITLE = "Efficient Generation of Speech Adversarial Examples with Generative
Model",
BOOKTITLE = IWDW20,
YEAR = "2020",
PAGES = "251-264",
BIBSOURCE = "http://www.visionbib.com/bibliography/other1024ss1.html#TT378915"}
@inproceedings{bb384835,
AUTHOR = "Zhou, H. and Liu, Z. and Xu, X. and Luo, P. and Wang, X.",
TITLE = "Vision-Infused Deep Audio Inpainting",
BOOKTITLE = ICCV19,
YEAR = "2019",
PAGES = "283-292",
BIBSOURCE = "http://www.visionbib.com/bibliography/other1024ss1.html#TT378916"}
@inproceedings{bb384836,
AUTHOR = "Bailer, W. and Wijnants, M. and Lievens, H. and Claes, S.",
TITLE = "Multimedia Analytics Challenges and Opportunities for Creating
Interactive Radio Content",
BOOKTITLE = MMMod20,
YEAR = "2020",
PAGES = "II:375-387",
BIBSOURCE = "http://www.visionbib.com/bibliography/other1024ss1.html#TT378917"}
@inproceedings{bb384837,
AUTHOR = "Huang, T. and Wang, H.X. and Chen, Y. and He, P.S.",
TITLE = "GRU-SVM Model for Synthetic Speech Detection",
BOOKTITLE = IWDW19,
YEAR = "2019",
PAGES = "115-125",
BIBSOURCE = "http://www.visionbib.com/bibliography/other1024ss1.html#TT378918"}
@inproceedings{bb384838,
AUTHOR = "Wong, A. and Xu, A. and Dudek, G.",
TITLE = "Investigating Trust Factors in Human-Robot Shared Control:
Implicit Gender Bias Around Robot Voice",
BOOKTITLE = CRV19,
YEAR = "2019",
PAGES = "195-200",
BIBSOURCE = "http://www.visionbib.com/bibliography/other1024ss1.html#TT378919"}
@inproceedings{bb384839,
AUTHOR = "Xiao, L. and Wang, Z.",
TITLE = "Dense Convolutional Recurrent Neural Network for Generalized Speech
Animation",
BOOKTITLE = ICPR18,
YEAR = "2018",
PAGES = "633-638",
BIBSOURCE = "http://www.visionbib.com/bibliography/other1024ss1.html#TT378920"}
@inproceedings{bb384840,
AUTHOR = "Shah, N.J. and Patil, H.A.",
TITLE = "Analysis of Features and Metrics for Alignment in Text-Dependent Voice
Conversion",
BOOKTITLE = PReMI17,
YEAR = "2017",
PAGES = "299-307",
BIBSOURCE = "http://www.visionbib.com/bibliography/other1024ss1.html#TT378921"}
@inproceedings{bb384841,
AUTHOR = "Rybarova, R. and Drozd, I. and Rozinaj, G.",
TITLE = "GUI for interactive speech synthesis",
BOOKTITLE = WSSIP16,
YEAR = "2016",
PAGES = "1-4",
BIBSOURCE = "http://www.visionbib.com/bibliography/other1024ss1.html#TT378922"}
@inproceedings{bb384842,
AUTHOR = "Coto Jimenez, M. and Goddard Close, J.",
TITLE = "LSTM Deep Neural Networks Postfiltering for Improving the Quality of
Synthetic Voices",
BOOKTITLE = MCPR16,
YEAR = "2016",
PAGES = "280-289",
BIBSOURCE = "http://www.visionbib.com/bibliography/other1024ss1.html#TT378923"}
@inproceedings{bb384843,
AUTHOR = "Vasek, M. and Rozinaj, G. and Rybarova, R.",
TITLE = "Letter-To-Sound conversion for speech synthesizer",
BOOKTITLE = WSSIP16,
YEAR = "2016",
PAGES = "1-4",
BIBSOURCE = "http://www.visionbib.com/bibliography/other1024ss1.html#TT378924"}
@inproceedings{bb384844,
AUTHOR = "Rybarova, R. and del Corral, G. and Rozinaj, G.",
TITLE = "Diphone spanish text-to-speech synthesizer",
BOOKTITLE = WSSIP15,
YEAR = "2015",
PAGES = "121-124",
BIBSOURCE = "http://www.visionbib.com/bibliography/other1024ss1.html#TT378925"}
@inproceedings{bb384845,
AUTHOR = "Verma, R. and Sarkar, P. and Rao, K.S.",
TITLE = "Conversion of neutral speech to storytelling style speech",
BOOKTITLE = ICAPR15,
YEAR = "2015",
PAGES = "1-6",
BIBSOURCE = "http://www.visionbib.com/bibliography/other1024ss1.html#TT378926"}
@inproceedings{bb384846,
AUTHOR = "Narendra, N.P. and Rao, K.S.",
TITLE = "Optimal residual frame based source modeling for HMM-based speech
synthesis",
BOOKTITLE = ICAPR15,
YEAR = "2015",
PAGES = "1-5",
BIBSOURCE = "http://www.visionbib.com/bibliography/other1024ss1.html#TT378927"}
@inproceedings{bb384847,
AUTHOR = "Wang, Y. and Tao, J.H. and Yang, M.H. and Li, Y.",
TITLE = "Extended Decision Tree with or Relationship for HMM-Based Speech
Synthesis",
BOOKTITLE = ACPR13,
YEAR = "2013",
PAGES = "225-229",
BIBSOURCE = "http://www.visionbib.com/bibliography/other1024ss1.html#TT378928"}
@inproceedings{bb384848,
AUTHOR = "Gao, L. and Yu, H.Z. and Zhang, J.H. and Fang, H.P.",
TITLE = "Research on HMM_based speech synthesis for Lhasa dialect",
BOOKTITLE = IASP11,
YEAR = "2011",
PAGES = "429-433",
BIBSOURCE = "http://www.visionbib.com/bibliography/other1024ss1.html#TT378929"}
@inproceedings{bb384849,
AUTHOR = "Chakraborty, R. and Garain, U.",
TITLE = "Role of Synthetically Generated Samples on Speech Recognition in a
Resource-Scarce Language",
BOOKTITLE = ICPR10,
YEAR = "2010",
PAGES = "1618-1621",
BIBSOURCE = "http://www.visionbib.com/bibliography/other1024ss1.html#TT378930"}
@inproceedings{bb384850,
AUTHOR = "Rao, K.S. and Maity, S. and Taru, A. and Koolagudi, S.G.",
TITLE = "Unit Selection Using Linguistic, Prosodic and Spectral Distance for
Developing Text-to-Speech System in Hindi",
BOOKTITLE = PReMI09,
YEAR = "2009",
PAGES = "531-536",
BIBSOURCE = "http://www.visionbib.com/bibliography/other1024ss1.html#TT378931"}
@inproceedings{bb384851,
AUTHOR = "Bahrampour, A. and Barkhoda, W. and Azami, B.Z.",
TITLE = "Implementation of Three Text to Speech Systems for Kurdish Language",
BOOKTITLE = CIARP09,
YEAR = "2009",
PAGES = "321-328",
BIBSOURCE = "http://www.visionbib.com/bibliography/other1024ss1.html#TT378932"}
@inproceedings{bb384852,
AUTHOR = "Shirbahadurkar, S.D. and Bormane, D.S.",
TITLE = "Marathi Language Speech Synthesizer Using Concatenative Synthesis
Strategy (Spoken in Maharashtra, India)",
BOOKTITLE = ICMV09,
YEAR = "2009",
PAGES = "181-185",
BIBSOURCE = "http://www.visionbib.com/bibliography/other1024ss1.html#TT378933"}
@inproceedings{bb384853,
AUTHOR = "Tuckova, J. and Holub, J. and Dubeda, T.",
TITLE = "Technical and Phonetic Aspects of Speech Quality Assessment:
The Case of Prosody Synthesis",
BOOKTITLE = COST08,
YEAR = "2008",
PAGES = "126-132",
BIBSOURCE = "http://www.visionbib.com/bibliography/other1024ss1.html#TT378934"}
@inproceedings{bb384854,
AUTHOR = "Bauer, D. and Kannampuzha, J. and Kroger, B.J.",
TITLE = "Articulatory Speech Re-synthesis:
Profiting from Natural Acoustic Speech Data",
BOOKTITLE = COST08,
YEAR = "2008",
PAGES = "344-355",
BIBSOURCE = "http://www.visionbib.com/bibliography/other1024ss1.html#TT378935"}
@inproceedings{bb384855,
AUTHOR = "Gu, H.Y. and Cai, C.L. and Cai, S.F.",
TITLE = "An HNM-Based Speaker-Nonspecific Timbre Transformation Scheme for
Speech Synthesis",
BOOKTITLE = CISP09,
YEAR = "2009",
PAGES = "1-5",
BIBSOURCE = "http://www.visionbib.com/bibliography/other1024ss1.html#TT378936"}
@article{bb384856,
AUTHOR = "Lung, S.Y. and Chen, C.C.T.",
TITLE = "A new approach for text-independent speaker recognition",
JOURNAL = PR,
VOLUME = "33",
YEAR = "2000",
NUMBER = "8",
MONTH = "August",
PAGES = "1401-1403",
BIBSOURCE = "http://www.visionbib.com/bibliography/other1024.html#TT378937"}
@article{bb384857,
AUTHOR = "Lung, S.Y.",
TITLE = "Multi-resolution form of SVD for text-independent speaker recognition",
JOURNAL = PR,
VOLUME = "35",
YEAR = "2002",
NUMBER = "7",
MONTH = "July",
PAGES = "1637-1639",
BIBSOURCE = "http://www.visionbib.com/bibliography/other1024.html#TT378938"}
@article{bb384858,
AUTHOR = "Lung, S.Y.",
TITLE = "Further reduced form of wavelet feature for text independent speaker
recognition",
JOURNAL = PR,
VOLUME = "37",
YEAR = "2004",
NUMBER = "7",
MONTH = "July",
PAGES = "1565-1566",
BIBSOURCE = "http://www.visionbib.com/bibliography/other1024.html#TT378939"}
@article{bb384859,
AUTHOR = "Lung, S.Y.",
TITLE = "Feature extracted from wavelet eigenfunction estimation for
text-independent speaker recognition",
JOURNAL = PR,
VOLUME = "37",
YEAR = "2004",
NUMBER = "7",
MONTH = "July",
PAGES = "1543-1544",
BIBSOURCE = "http://www.visionbib.com/bibliography/other1024.html#TT378940"}
@article{bb384860,
AUTHOR = "Lung, S.Y.",
TITLE = "Wavelet feature domain adaptive noise reduction using learning
algorithm for text-independent speaker recognition",
JOURNAL = PR,
VOLUME = "40",
YEAR = "2007",
NUMBER = "9",
MONTH = "September",
PAGES = "2603-2606",
BIBSOURCE = "http://www.visionbib.com/bibliography/other1024.html#TT378941"}
@article{bb384861,
AUTHOR = "Lung, S.Y.",
TITLE = "Efficient text independent speaker recognition with wavelet feature
selection based multilayered neural network using supervised learning
algorithm",
JOURNAL = PR,
VOLUME = "40",
YEAR = "2007",
NUMBER = "12",
MONTH = "December",
PAGES = "3616-3620",
BIBSOURCE = "http://www.visionbib.com/bibliography/other1024.html#TT378942"}
@article{bb384862,
AUTHOR = "Lung, S.Y.",
TITLE = "Distributed genetic algorithm for Gaussian mixture model based speaker
identification",
JOURNAL = PR,
VOLUME = "36",
YEAR = "2003",
NUMBER = "10",
MONTH = "October",
PAGES = "2479-2481",
BIBSOURCE = "http://www.visionbib.com/bibliography/other1024.html#TT378943"}
@article{bb384863,
AUTHOR = "Lung, S.Y.",
TITLE = "Adaptive fuzzy wavelet algorithm for text-independent speaker
recognition",
JOURNAL = PR,
VOLUME = "37",
YEAR = "2004",
NUMBER = "10",
MONTH = "October",
PAGES = "2095-2096",
BIBSOURCE = "http://www.visionbib.com/bibliography/other1024.html#TT378944"}
@article{bb384864,
AUTHOR = "Lung, S.Y.",
TITLE = "Wavelet feature selection based neural networks with application to the
text independent speaker identification",
JOURNAL = PR,
VOLUME = "39",
YEAR = "2006",
NUMBER = "8",
MONTH = "August",
PAGES = "1518-1521",
BIBSOURCE = "http://www.visionbib.com/bibliography/other1024.html#TT378945"}
@article{bb384865,
AUTHOR = "Lung, S.Y.",
TITLE = "Feature extracted from wavelet decomposition using biorthogonal Riesz
basis for text-independent speaker recognition",
JOURNAL = PR,
VOLUME = "41",
YEAR = "2008",
NUMBER = "10",
MONTH = "October",
PAGES = "3068-3070",
BIBSOURCE = "http://www.visionbib.com/bibliography/other1024.html#TT378946"}
@article{bb384866,
AUTHOR = "Chen, K. and Wu, T.Y. and Zhang, H.J.",
TITLE = "On the use of nearest feature line for speaker identification",
JOURNAL = PRL,
VOLUME = "23",
YEAR = "2002",
NUMBER = "14",
MONTH = "December",
PAGES = "1735-1746",
BIBSOURCE = "http://www.visionbib.com/bibliography/other1024.html#TT378947"}
@article{bb384867,
AUTHOR = "Ramachandran, R.P. and Farrell, K.R. and Ramachandran, R. and Mammone, R.J.",
TITLE = "Speaker recognition:
general classifier approaches and data fusion methods",
JOURNAL = PR,
VOLUME = "35",
YEAR = "2002",
NUMBER = "12",
MONTH = "December",
PAGES = "2801-2821",
BIBSOURCE = "http://www.visionbib.com/bibliography/other1024.html#TT378948"}
@article{bb384868,
AUTHOR = "Chen, K.",
TITLE = "Towards better making a decision in speaker verification",
JOURNAL = PR,
VOLUME = "36",
YEAR = "2003",
NUMBER = "2",
MONTH = "February",
PAGES = "329-346",
BIBSOURCE = "http://www.visionbib.com/bibliography/other1024.html#TT378949"}
@article{bb384869,
AUTHOR = "Rodriguez Linares, L. and Garcia Mateo, C. and Alba Castro, J.L.",
TITLE = "On combining classifiers for speaker authentication",
JOURNAL = PR,
VOLUME = "36",
YEAR = "2003",
NUMBER = "2",
MONTH = "February",
PAGES = "347-359",
BIBSOURCE = "http://www.visionbib.com/bibliography/other1024.html#TT378950"}
@article{bb384870,
AUTHOR = "Damper, R.I. and Higgins, J.E.",
TITLE = "Improving speaker identification in noise by subband processing and
decision fusion",
JOURNAL = PRL,
VOLUME = "24",
YEAR = "2003",
NUMBER = "13",
MONTH = "September",
PAGES = "2167-2173",
BIBSOURCE = "http://www.visionbib.com/bibliography/other1024.html#TT378951"}
@article{bb384871,
AUTHOR = "Besacier, L. and Mayorga, P. and Bonastre, J.F. and Fredouille, C. and Meignier, S.",
TITLE = "Overview of compression and packet loss effects in speech biometrics",
JOURNAL = VISP,
VOLUME = "150",
YEAR = "2003",
NUMBER = "6",
MONTH = "December",
PAGES = "372-376",
BIBSOURCE = "http://www.visionbib.com/bibliography/other1024.html#TT378952"}
@inproceedings{bb384872,
AUTHOR = "Besacier, L. and Bonastre, J.F.",
TITLE = "Time and frequency pruning for speaker identification",
BOOKTITLE = ICPR98,
YEAR = "1998",
PAGES = "Vol II: 1619-1621",
BIBSOURCE = "http://www.visionbib.com/bibliography/other1024.html#TT378953"}
@article{bb384873,
AUTHOR = "Rodriguez Linares, L. and Garcia Mateo, C.",
TITLE = "Application of fusion techniques to speaker authentication over ip
networks",
JOURNAL = VISP,
VOLUME = "150",
YEAR = "2003",
NUMBER = "6",
MONTH = "December",
PAGES = "377-382",
BIBSOURCE = "http://www.visionbib.com/bibliography/other1024.html#TT378954"}
@article{bb384874,
AUTHOR = "Chen, C.C.T. and Chen, C.T. and Hou, C.K.",
TITLE = "Speaker identification using hybrid Karhunen-Loeve transform and
Gaussian mixture model approach",
JOURNAL = PR,
VOLUME = "37",
YEAR = "2004",
NUMBER = "5",
MONTH = "May",
PAGES = "1073-1075",
BIBSOURCE = "http://www.visionbib.com/bibliography/other1024.html#TT378955"}
@article{bb384875,
AUTHOR = "Lee, K.Y.",
TITLE = "Local fuzzy PCA based GMM with dimension reduction on speaker
identification",
JOURNAL = PRL,
VOLUME = "25",
YEAR = "2004",
NUMBER = "16",
MONTH = "December",
PAGES = "1811-1817",
BIBSOURCE = "http://www.visionbib.com/bibliography/other1024.html#TT378956"}
@article{bb384876,
AUTHOR = "Mashao, D.J. and Skosan, M.",
TITLE = "Combining classifier decisions for robust speaker identification",
JOURNAL = PR,
VOLUME = "39",
YEAR = "2006",
NUMBER = "1",
MONTH = "January",
PAGES = "147-155",
BIBSOURCE = "http://www.visionbib.com/bibliography/other1024.html#TT378957"}
@article{bb384877,
AUTHOR = "Skosan, M. and Mashao, D.J.",
TITLE = "Modified Segmental Histogram Equalization for robust speaker
verification",
JOURNAL = PRL,
VOLUME = "27",
YEAR = "2006",
NUMBER = "5",
MONTH = "April",
PAGES = "479-486",
BIBSOURCE = "http://www.visionbib.com/bibliography/other1024.html#TT378958"}
@article{bb384878,
AUTHOR = "Ariyaeeinia, A.M. and Fortuna, J. and Sivakumaran, P. and Malegaonkar, A.",
TITLE = "Verification effectiveness in open-set speaker identification",
JOURNAL = VISP,
VOLUME = "153",
YEAR = "2006",
NUMBER = "5",
MONTH = "October",
PAGES = "618-624",
BIBSOURCE = "http://www.visionbib.com/bibliography/other1024.html#TT378959"}
@article{bb384879,
AUTHOR = "Zhou, G. and Mikhael, W.B.",
TITLE = "Speaker identification based on adaptive discriminative vector
quantisation",
JOURNAL = VISP,
VOLUME = "153",
YEAR = "2006",
NUMBER = "6",
MONTH = "December",
PAGES = "754-760",
BIBSOURCE = "http://www.visionbib.com/bibliography/other1024.html#TT378960"}
@article{bb384880,
AUTHOR = "Park, C.M. and Thapa, D. and Wang, G.N.",
TITLE = "Speech authentication system using digital watermarking and pattern
recovery",
JOURNAL = PRL,
VOLUME = "28",
YEAR = "2007",
NUMBER = "8",
MONTH = "June",
PAGES = "931-938",
BIBSOURCE = "http://www.visionbib.com/bibliography/other1024.html#TT378961"}
@article{bb384881,
AUTHOR = "Faundez Zanuy, M. and Hagmuller, M. and Kubin, G.",
TITLE = "Speaker identification security improvement by means of speech
watermarking",
JOURNAL = PR,
VOLUME = "40",
YEAR = "2007",
NUMBER = "11",
MONTH = "November",
PAGES = "3027-3034",
BIBSOURCE = "http://www.visionbib.com/bibliography/other1024.html#TT378962"}
@article{bb384882,
AUTHOR = "Chetouani, M. and Faundez Zanuy, M. and Gas, B. and Zarader, J.L.",
TITLE = "Investigation on LP-residual representations for speaker identification",
JOURNAL = PR,
VOLUME = "42",
YEAR = "2009",
NUMBER = "3",
MONTH = "March",
PAGES = "487-494",
BIBSOURCE = "http://www.visionbib.com/bibliography/other1024.html#TT378963"}
@article{bb384883,
AUTHOR = "Kinnunen, T. and Saastamoinen, J. and Hautamaki, V. and Vinni, M. and Franti, P.",
TITLE = "Comparative evaluation of maximum a Posteriori vector quantization and
Gaussian mixture models in speaker verification",
JOURNAL = PRL,
VOLUME = "30",
YEAR = "2009",
NUMBER = "4",
MONTH = "March",
PAGES = "341-347",
BIBSOURCE = "http://www.visionbib.com/bibliography/other1024.html#TT378964"}
@article{bb384884,
AUTHOR = "Chao, Y.H. and Tsai, W.H. and Wang, H.M. and Chang, R.C.",
TITLE = "Improving the characterization of the alternative hypothesis via
minimum verification error training with applications to speaker
verification",
JOURNAL = PR,
VOLUME = "42",
YEAR = "2009",
NUMBER = "7",
MONTH = "July",
PAGES = "1351-1360",
BIBSOURCE = "http://www.visionbib.com/bibliography/other1024.html#TT378965"}
@article{bb384885,
AUTHOR = "Temko, A. and Nadeu, C.",
TITLE = "Acoustic event detection in meeting-room environments",
JOURNAL = PRL,
VOLUME = "30",
YEAR = "2009",
NUMBER = "14",
MONTH = "October",
PAGES = "1281-1288",
BIBSOURCE = "http://www.visionbib.com/bibliography/other1024.html#TT378966"}
@article{bb384886,
AUTHOR = "Kim, S. and Ji, M.Y. and Kim, H.",
TITLE = "Robust speaker recognition based on filtering in autocorrelation domain
and sub-band feature recombination",
JOURNAL = PRL,
VOLUME = "31",
YEAR = "2010",
NUMBER = "7",
MONTH = "May",
PAGES = "593-599",
BIBSOURCE = "http://www.visionbib.com/bibliography/other1024.html#TT378967"}
@article{bb384887,
AUTHOR = "Li, H. and Ma, B. and Lee, K.A.",
TITLE = "Spoken Language Recognition: From Fundamentals to Practice",
JOURNAL = PIEEE,
VOLUME = "100",
YEAR = "2013",
NUMBER = "5",
MONTH = "May",
PAGES = "1136-1159",
BIBSOURCE = "http://www.visionbib.com/bibliography/other1024.html#TT378968"}
@article{bb384888,
AUTHOR = "Li, H. and Ma, B.",
TITLE = "TechWare: Speaker and Spoken Language Recognition Resources",
JOURNAL = SPMag,
VOLUME = "27",
YEAR = "2010",
NUMBER = "6",
PAGES = "139-142",
BIBSOURCE = "http://www.visionbib.com/bibliography/other1024.html#TT378969"}
@article{bb384889,
AUTHOR = "Ajmera, P.K. and Jadhav, D.V. and Holambe, R.S.",
TITLE = "Text-independent speaker identification using Radon and discrete cosine
transforms based features from speech spectrogram",
JOURNAL = PR,
VOLUME = "44",
YEAR = "2011",
NUMBER = "10-11",
MONTH = "October",
PAGES = "2749-2759",
BIBSOURCE = "http://www.visionbib.com/bibliography/other1024.html#TT378970"}
@article{bb384890,
AUTHOR = "Kinnunen, T. and Sidoroff, I. and Tuononen, M. and Franti, P.",
TITLE = "Comparison of clustering methods:
A case study of text-independent speaker modeling",
JOURNAL = PRL,
VOLUME = "32",
YEAR = "2011",
NUMBER = "13",
MONTH = "October",
PAGES = "1604-1617",
BIBSOURCE = "http://www.visionbib.com/bibliography/other1024.html#TT378971"}
@inproceedings{bb384891,
AUTHOR = "Kinnunen, T. and Karpov, E. and Franti, P.",
TITLE = "A Speaker Pruning Algorithm for Real-Time Speaker Identification",
BOOKTITLE = AVBPA03,
YEAR = "2003",
PAGES = "639-646",
BIBSOURCE = "http://www.visionbib.com/bibliography/other1024.html#TT378972"}
@inproceedings{bb384892,
AUTHOR = "Kinnunen, T. and Franti, P.",
TITLE = "Speaker Discriminative Weighting Method for VQ-Based Speaker
Identification",
BOOKTITLE = AVBPA01,
YEAR = "2001",
PAGES = "150",
BIBSOURCE = "http://www.visionbib.com/bibliography/other1024.html#TT378973"}
@article{bb384893,
AUTHOR = "Zao, L. and Coelho, R.",
TITLE = "Colored Noise Based Multicondition Training Technique for Robust
Speaker Identification",
JOURNAL = SPLetters,
VOLUME = "18",
YEAR = "2011",
NUMBER = "11",
MONTH = "November",
PAGES = "675-678",
BIBSOURCE = "http://www.visionbib.com/bibliography/other1024.html#TT378974"}
@article{bb384894,
AUTHOR = "Hanilci, C. and Kinnunen, T. and Ertas, F. and Saeidi, R. and Pohjalainen, J. and Alku, P.",
TITLE = "Regularized All-Pole Models for Speaker Verification Under Noisy
Environments",
JOURNAL = SPLetters,
VOLUME = "19",
YEAR = "2012",
NUMBER = "3",
MONTH = "March",
PAGES = "163-166",
BIBSOURCE = "http://www.visionbib.com/bibliography/other1024.html#TT378975"}
@article{bb384895,
AUTHOR = "Salamin, H. and Vinciarelli, A.",
TITLE = "Automatic Role Recognition in Multiparty Conversations: An Approach
Based on Turn Organization, Prosody, and Conditional Random Fields",
JOURNAL = MultMed,
VOLUME = "14",
YEAR = "2012",
NUMBER = "2",
PAGES = "338-345",
BIBSOURCE = "http://www.visionbib.com/bibliography/other1024.html#TT378976"}
@article{bb384896,
AUTHOR = "Tang, H. and Chu, S. and Hasegawa Johnson, M. and Huang, T.S.",
TITLE = "Partially Supervised Speaker Clustering",
JOURNAL = PAMI,
VOLUME = "34",
YEAR = "2012",
NUMBER = "5",
MONTH = "May",
PAGES = "959-971",
BIBSOURCE = "http://www.visionbib.com/bibliography/other1024.html#TT378977"}
@article{bb384897,
AUTHOR = "Montalvao, J. and Araujo, M.R.R.",
TITLE = "Is masking a relevant aspect lacking in MFCC? A speaker verification
perspective",
JOURNAL = PRL,
VOLUME = "33",
YEAR = "2012",
NUMBER = "16",
MONTH = "December",
PAGES = "2156-2165",
BIBSOURCE = "http://www.visionbib.com/bibliography/other1024.html#TT378978"}
@article{bb384898,
AUTHOR = "Garimella, S. and Mallidi, S.H. and Hermansky, H.",
TITLE = "Regularized Auto-Associative Neural Networks for Speaker Verification",
JOURNAL = SPLetters,
VOLUME = "19",
YEAR = "2012",
NUMBER = "12",
MONTH = "December",
PAGES = "841-844",
BIBSOURCE = "http://www.visionbib.com/bibliography/other1024.html#TT378979"}
@article{bb384899,
AUTHOR = "Sahidullah, M. and Saha, G.",
TITLE = "A Novel Windowing Technique for Efficient Computation of MFCC for
Speaker Recognition",
JOURNAL = SPLetters,
VOLUME = "20",
YEAR = "2013",
NUMBER = "2",
MONTH = "February",
PAGES = "149-152",
BIBSOURCE = "http://www.visionbib.com/bibliography/other1024.html#TT378980"}
Last update:Aug 19, 2026 at 13:26:35