{"generated_at":"2026-10-10T14:00:35Z","total":259,"page":0,"pages":3,"page_size":100,"next":"/api/catalog?topic=speech-audio&page=1","documents":[{"id":"doi-776691167eee8ab3fe1d","title":"Karelian speech recognition system with support for Karelian-Russian code-switching","year":2026,"authors":["I. S. Kipyatkova","M. D. Dolgushin","K. O. Kiseleva"],"author_count":4,"journal":"Научно-технический вестник информационных технологий, механики и оптики","doi":"10.17586/2226-1494-2026-26-4-826-834","source":"doaj","topics":["speech-audio"],"keywords":["автоматическое распознавание речи","переключение кодов","карельский язык","малоресурсные языки","аугментация данных","внутрисловное переключение кода","wav2vec2-bert 2.0","языковое моделирование"],"license":"CC0-1.0","has_fulltext":false,"has_references":false,"rights_class":"metadata_only","access":"paid","version":1,"content_endpoint":"/api/document/doi-776691167eee8ab3fe1d","fulltext_endpoint":null,"references_endpoint":null},{"id":"doi-1e47c877464a7f2a4af1","title":"SONIVA database: Speech recognition validation in aphasia","year":2026,"authors":["Giulia Sanguedolce","Cathy J. Price","Sophie Brook"],"author_count":7,"journal":"Scientific Data","doi":"10.1038/s41597-026-07596-3","source":"doaj","topics":["databases-data-management","speech-audio"],"keywords":["Science"],"license":"CC0-1.0","has_fulltext":false,"has_references":false,"rights_class":"metadata_only","access":"paid","version":1,"content_endpoint":"/api/document/doi-1e47c877464a7f2a4af1","fulltext_endpoint":null,"references_endpoint":null},{"id":"doi-d5085a6f693300c5a907","title":"From Single Modality to Multimodality in Dysarthric Speech Recognition: A Systematic Scoping Review","year":2026,"authors":["Ehsan Eslami","Hossein Karshenas","Alireza Darvishy"],"author_count":4,"journal":"IEEE Access","doi":"10.1109/access.2026.3737868","source":"doaj","topics":["speech-audio"],"keywords":["Dysarthric speech recognition","multimodal ASR","audio-visual fusion","articulatory features","deep learning","Electrical engineering. Electronics. Nuclear engineering"],"license":"CC0-1.0","has_fulltext":false,"has_references":false,"rights_class":"metadata_only","access":"paid","version":1,"content_endpoint":"/api/document/doi-d5085a6f693300c5a907","fulltext_endpoint":null,"references_endpoint":null},{"id":"doi-8ab4b3548acd7c4a5f7b","title":"Encoder-guided conditional CNN-WGAN-GP for emotion-controlled speech synthesis with neural vocoder reconstruction and objective quality evaluation","year":2026,"authors":["Dharmaiah Devarapalli","Koduri Abhiram","Chandu Venkata Sai Phani Gopi"],"author_count":6,"journal":"Array","doi":"10.1016/j.array.2026.101258","source":"doaj","topics":["speech-audio"],"keywords":["Emotion-aware speech synthesis","Encoder-guided GAN","WGAN-GP","Mel-spectrogram generation","Emotional speech generation","HiFi-GAN","Computer engineering. Computer hardware","Electronic computers. Computer science"],"license":"CC0-1.0","has_fulltext":false,"has_references":false,"rights_class":"metadata_only","access":"paid","version":1,"content_endpoint":"/api/document/doi-8ab4b3548acd7c4a5f7b","fulltext_endpoint":null,"references_endpoint":null},{"id":"doi-4a20cb0148ef3f6ad2b0","title":"Impact of migrating from paired to sequential stimulation on speech recognition in cochlear implant users","year":2026,"authors":["Aline Faria de Sousa","Aline Santos","Lucas Bevilacqua Alves da Costa"],"author_count":5,"journal":"Brazilian Journal of Otorhinolaryngology","doi":"10.1016/j.bjorl.2026.101910","source":"doaj","topics":["speech-audio"],"keywords":["Cochlear implants","Auditory perception","Hearing loss","Speech perception","Otorhinolaryngology"],"license":"CC0-1.0","has_fulltext":false,"has_references":false,"rights_class":"metadata_only","access":"paid","version":1,"content_endpoint":"/api/document/doi-4a20cb0148ef3f6ad2b0","fulltext_endpoint":null,"references_endpoint":null},{"id":"doi-188dfb569f37277fec5f","title":"Emotion-Robust Speaker Recognition Through Synthetic Emotional Spectrogram Augmentation","year":2026,"authors":["Eva Jakubcova","Maros Jakubec"],"author_count":2,"journal":"Information","doi":"10.3390/info17090851","source":"doaj","topics":["speech-audio"],"keywords":["speaker recognition","emotional speech","data augmentation","emotional voice conversion","mel-spectrogram","generative adversarial network","Information technology"],"license":"CC0-1.0","has_fulltext":false,"has_references":false,"rights_class":"metadata_only","access":"paid","version":1,"content_endpoint":"/api/document/doi-188dfb569f37277fec5f","fulltext_endpoint":null,"references_endpoint":null},{"id":"doi-c5fa990da5b3335df3f3","title":"Laryngeal-speech recognition system based on graphene nanocrystalline carbon film sensing","year":2026,"authors":["Wang Kai","Ma Conghui","Chai Shanglei"],"author_count":7,"journal":"Shenzhen Daxue xuebao. Ligong ban","doi":"10.3724/sp.j.1249.2026.05622","source":"doaj","topics":["speech-audio"],"keywords":["opto-mechatronics engineering and applications","laryngeal-speech recognition system","electron cyclotron resonance carbon film","flexible sensor","deep learning","pattern prediction","Technology"],"license":"CC0-1.0","has_fulltext":false,"has_references":false,"rights_class":"metadata_only","access":"paid","version":1,"content_endpoint":"/api/document/doi-c5fa990da5b3335df3f3","fulltext_endpoint":null,"references_endpoint":null},{"id":"doi-fdf54e2c76c0e076400d","title":"Multi-Scale Residual Attention Network for Chinese Speech Recognition Through Collaborative Design","year":2026,"authors":["Yuying Li","Hongjie Wan","Akash Sutradhar"],"author_count":4,"journal":"Entropy","doi":"10.3390/e28091037","source":"doaj","topics":["speech-audio"],"keywords":["Chinese speech recognition","multi-scale feature modeling","residual convolutional network","attention mechanism","deep learning","ASR","Science","Astrophysics"],"license":"CC0-1.0","has_fulltext":false,"has_references":false,"rights_class":"metadata_only","access":"paid","version":1,"content_endpoint":"/api/document/doi-fdf54e2c76c0e076400d","fulltext_endpoint":null,"references_endpoint":null},{"id":"doi-86e41c569a7e1eb31d01","title":"Real-Time Granular Audio Processing Using Raspberry Pi and Digital Signal Processing Algorithms","year":2026,"authors":["Richard Lorenzo R. Lising","Meo Vincent C. Caya"],"author_count":2,"journal":"Engineering Proceedings","doi":"10.3390/engproc2026134097","source":"doaj","topics":["speech-audio"],"keywords":["real-time processing","granular synthesis","Raspberry Pi","digital audio","digital signal processing","Engineering machinery, tools, and implements"],"license":"CC0-1.0","has_fulltext":false,"has_references":false,"rights_class":"metadata_only","access":"paid","version":1,"content_endpoint":"/api/document/doi-86e41c569a7e1eb31d01","fulltext_endpoint":null,"references_endpoint":null},{"id":"doi-ec08f8425a7899946324","title":"Leveraging MFCC and Mel-Spectrogram Representations for Deep Learning-Based Speech Recognition","year":2026,"authors":["Jose Antonio Lopez-Olvera","Hector Manuel Perez-Meana","Elizabeth Garcia-Rios"],"author_count":4,"journal":"Engineering Proceedings","doi":"10.3390/engproc2026123022","source":"doaj","topics":["speech-audio"],"keywords":["MFCC","mel-Spectrogram","hamming window","CNN","Engineering machinery, tools, and implements"],"license":"CC0-1.0","has_fulltext":false,"has_references":false,"rights_class":"metadata_only","access":"paid","version":1,"content_endpoint":"/api/document/doi-ec08f8425a7899946324","fulltext_endpoint":null,"references_endpoint":null},{"id":"doi-6d262303d57b6c8440dc","title":"Research on an Enhanced Zipformer With Multi-Feature Attention for Aviation Speech Recognition","year":2026,"authors":["Geng Yu","Jianxing Liang","Lihua Wen"],"author_count":5,"journal":"IEEE Access","doi":"10.1109/access.2026.3733308","source":"doaj","topics":["speech-audio"],"keywords":["Automatic speech recognition","air traffic control","end-to-end","Zipformer","multi-scale parallel convolution","multi-attention","Electrical engineering. Electronics. Nuclear engineering"],"license":"CC0-1.0","has_fulltext":false,"has_references":false,"rights_class":"metadata_only","access":"paid","version":1,"content_endpoint":"/api/document/doi-6d262303d57b6c8440dc","fulltext_endpoint":null,"references_endpoint":null},{"id":"doi-81bf49f1d5ccad7e522e","title":"The Influence of Prosodic-Acoustic Features in Automatic Speech Recognition of Two Inland Brazilian Dialects: A Pilot Study","year":2026,"authors":["Leônidas Silva Jr.","Luciani Ester Tenani","João Marcelo Monte"],"author_count":3,"journal":"Cadernos de Linguística","doi":"10.25189/2675-4916.2026.v7.n7.id1016","source":"doaj","topics":["speech-audio"],"keywords":["Prosodic-Acoustic Features","Dialect Classification","Automatic Speech Recognition","Brazilian Portuguese","Sociophonetics","Philology. Linguistics"],"license":"CC0-1.0","has_fulltext":false,"has_references":false,"rights_class":"metadata_only","access":"paid","version":1,"content_endpoint":"/api/document/doi-81bf49f1d5ccad7e522e","fulltext_endpoint":null,"references_endpoint":null},{"id":"doi-444218ead18169299731","title":"Soft Active Electromyography Interface for Machine Learning‐Enabled Silent Speech Recognition","year":2026,"authors":["Yuta Kurotaki","Shusuke Yamakoshi","Reitaro Yoshida"],"author_count":9,"journal":"Advanced Intelligent Systems","doi":"10.1002/aisy.70440","source":"doaj","topics":["speech-audio"],"keywords":["deep neural networks","human–machine interfaces","liquid metals","silent speech interfaces","soft sensors","wearable electronics","Computer engineering. Computer hardware","Control engineering systems. Automatic machinery (General)"],"license":"CC0-1.0","has_fulltext":false,"has_references":false,"rights_class":"metadata_only","access":"paid","version":1,"content_endpoint":"/api/document/doi-444218ead18169299731","fulltext_endpoint":null,"references_endpoint":null},{"id":"doi-78fd3d6e62e4dfc59739","title":"Factors Affecting Visual-Only Speech Recognition in Individuals With Cochlear Implants","year":2026,"authors":["Zhikai Zhang MD, PhD","Chaogang Wei MD","Keli Cao MD, PhD"],"author_count":4,"journal":"Ear, Nose & Throat Journal","doi":"10.1177/01455613241234821","source":"doaj","topics":["speech-audio"],"keywords":["Otorhinolaryngology"],"license":"CC0-1.0","has_fulltext":false,"has_references":false,"rights_class":"metadata_only","access":"paid","version":1,"content_endpoint":"/api/document/doi-78fd3d6e62e4dfc59739","fulltext_endpoint":null,"references_endpoint":null},{"id":"doi-0bf779ce272fc1d9e3d1","title":"INESCO Dataset: indonesian expressive speech corpus for emotion speech synthesis","year":2026,"authors":["Elok Anggrayni","Agus Muhamad Hatta","Dhany Arifianto"],"author_count":5,"journal":"Data in Brief","doi":"10.1016/j.dib.2026.113058","source":"doaj","topics":["speech-audio"],"keywords":["Expressive speech synthesis","Indonesian corpus","Low-resource language","Natural language processing","Signal processing","Speech dataset","Computer applications to medicine. Medical informatics","Science (General)"],"license":"CC0-1.0","has_fulltext":false,"has_references":false,"rights_class":"metadata_only","access":"paid","version":1,"content_endpoint":"/api/document/doi-0bf779ce272fc1d9e3d1","fulltext_endpoint":null,"references_endpoint":null},{"id":"doi-00e2a8aa120677057cfc","title":"Optimizing Whisper models for Turkish automatic speech recognition: A comprehensive study on parameter-efficient fine-tuning and model compression","year":2026,"authors":["Saadin Oyucu","Bilgehan Arslan","Cemal Kocak"],"author_count":4,"journal":"Engineering Science and Technology, an International Journal","doi":"10.1016/j.jestch.2026.102505","source":"doaj","topics":["speech-audio"],"keywords":["Turkish speech recognition","Whisper fine-tuning","LoRA adaptation","Model quantization","Decoder specialization","CTranslate2","Engineering (General). Civil engineering (General)"],"license":"CC0-1.0","has_fulltext":false,"has_references":false,"rights_class":"metadata_only","access":"paid","version":1,"content_endpoint":"/api/document/doi-00e2a8aa120677057cfc","fulltext_endpoint":null,"references_endpoint":null},{"id":"doi-d875e9fda738ec187e3e","title":"KAZMORPHLM: MORPHEME-AWARE LANGUAGE MODEL FOR KAZAKH AUTOMATIC SPEECH RECOGNITION","year":2026,"authors":["Yerlan Karabaliyev","Kateryna  Kolesnikova"],"author_count":2,"journal":"Scientific Journal of Astana IT University","doi":"10.37943/25scdm2312","source":"doaj","topics":["speech-audio"],"keywords":["morpheme language model","Kazakh speech recognition","agglutinative morphology","vowel harmony","morpheme segmentation","n-gram interpolation","Information technology"],"license":"CC0-1.0","has_fulltext":false,"has_references":false,"rights_class":"metadata_only","access":"paid","version":1,"content_endpoint":"/api/document/doi-d875e9fda738ec187e3e","fulltext_endpoint":null,"references_endpoint":null},{"id":"doi-aca32d6f2405d09b943f","title":"Hey AI: Can you trigger me? A comparison of Text-to-Speech synthesis and human speech in a virtual social stress paradigm","year":2026,"authors":["Sarah Roßkopf","Leon O.H. Kroczek","Theresa F. Wechsler"],"author_count":7,"journal":"Computers in Human Behavior: Artificial Humans","doi":"10.1016/j.chbah.2026.100371","source":"doaj","topics":["speech-audio"],"keywords":["Virtual social interactions","Text-to-Speech","Presence","Virtual reality","Stress","Physiological reactions","Electronic computers. Computer science","Information technology"],"license":"CC0-1.0","has_fulltext":false,"has_references":false,"rights_class":"metadata_only","access":"paid","version":1,"content_endpoint":"/api/document/doi-aca32d6f2405d09b943f","fulltext_endpoint":null,"references_endpoint":null},{"id":"doi-d5e80229beb79372901f","title":"Influence of factors related to electrode array placement on the early speech recognition of cochlear implant recipients with and without functional hearing preservation","year":2026,"authors":["Margaret T. Dillon","Madison E. Broome","Madison E. Broome"],"author_count":12,"journal":"Frontiers in Audiology and Otology","doi":"10.3389/fauot.2026.1908993","source":"doaj","topics":["speech-audio"],"keywords":["cochlear implant","electric-acoustic stimulation","electrode array placement","frequency-to-place mismatch","hearing preservation","spectral mismatch","Medicine"],"license":"CC0-1.0","has_fulltext":false,"has_references":false,"rights_class":"metadata_only","access":"paid","version":1,"content_endpoint":"/api/document/doi-d5e80229beb79372901f","fulltext_endpoint":null,"references_endpoint":null},{"id":"doi-a7b5d2e7d77dab69b36b","title":"HeyJay! A corpus of atypical speech for spoken language understanding and automatic speech recognition","year":2026,"authors":["Laureano Moro-Velazquez","Helin Wang","Alison Gunzler"],"author_count":7,"journal":"Scientific Data","doi":"10.1038/s41597-026-07497-5","source":"doaj","topics":["speech-audio"],"keywords":["Science"],"license":"CC0-1.0","has_fulltext":false,"has_references":false,"rights_class":"metadata_only","access":"paid","version":1,"content_endpoint":"/api/document/doi-a7b5d2e7d77dab69b36b","fulltext_endpoint":null,"references_endpoint":null},{"id":"doi-8d8755ed7d39cb940287","title":"Influence of hearing in the non-implanted ear and age on long-term speech recognition benefit for adult cochlear implant users with asymmetric hearing loss","year":2026,"authors":["Sylvia Mihailescu","Sylvia Mihailescu","Samantha P. Scharf"],"author_count":9,"journal":"Frontiers in Neuroscience","doi":"10.3389/fnins.2026.1871719","source":"doaj","topics":["speech-audio"],"keywords":["bimodal","cochlear implant","single-sided deafness","spatial hearing","spatial release from masking (SRM)","unilateral hearing loss (UHL)","Neurosciences. Biological psychiatry. Neuropsychiatry"],"license":"CC0-1.0","has_fulltext":false,"has_references":false,"rights_class":"metadata_only","access":"paid","version":1,"content_endpoint":"/api/document/doi-8d8755ed7d39cb940287","fulltext_endpoint":null,"references_endpoint":null},{"id":"doi-4fbee1679e302bf04fc1","title":"Reinforcement learning from human feedback improves automatic speech recognition models for ancient Thai language","year":2026,"authors":["Jettasic Popun","Wilaiporn Lee","Kanabadee Srisomboon"],"author_count":5,"journal":"Scientific Reports","doi":"10.1038/s41598-026-54756-x","source":"doaj","topics":["speech-audio"],"keywords":["Medicine","Science"],"license":"CC0-1.0","has_fulltext":false,"has_references":false,"rights_class":"metadata_only","access":"paid","version":1,"content_endpoint":"/api/document/doi-4fbee1679e302bf04fc1","fulltext_endpoint":null,"references_endpoint":null},{"id":"doi-177bd04547b8c6fd9f4f","title":"Persona-ASR: Bilingual Target-Speaker Speech Recognition for Kazakh–English Overlapping Speech","year":2026,"authors":["Rakhat Meiramov","Tomiris Rakhimzhanova","Adil Taibassarov"],"author_count":5,"journal":"Machine Learning and Knowledge Extraction","doi":"10.3390/make8080246","source":"doaj","topics":["speech-audio"],"keywords":["target-speaker ASR","multilingual speech recognition","Kazakh language","low-resource languages","WavLM","Computer engineering. Computer hardware"],"license":"CC0-1.0","has_fulltext":false,"has_references":false,"rights_class":"metadata_only","access":"paid","version":1,"content_endpoint":"/api/document/doi-177bd04547b8c6fd9f4f","fulltext_endpoint":null,"references_endpoint":null},{"id":"doi-0b3da9d4ee89d4015e3b","title":"Developing a Kazakh Audio–Visual Multimodal Speech Recognition Model Based on Hierarchical and Cross-Modal Attention","year":2026,"authors":["Turdybek Kurmetkan","Orken Mamyrbayev","Adem Tekerek"],"author_count":4,"journal":"Information","doi":"10.3390/info17080756","source":"doaj","topics":["speech-audio"],"keywords":["AVSR","Kazakh language","multimodal speech recognition","HuBERT","Vision Transformer","hierarchical attention","Information technology"],"license":"CC0-1.0","has_fulltext":false,"has_references":false,"rights_class":"metadata_only","access":"paid","version":1,"content_endpoint":"/api/document/doi-0b3da9d4ee89d4015e3b","fulltext_endpoint":null,"references_endpoint":null},{"id":"doi-01550a7b18e06d1fdb29","title":"Effects of Listener Position on Speech Recognition in a Simulated Multitalker Environment","year":2026,"authors":["William J. Bologna","Courtney King","Katie Esser"],"author_count":5,"journal":"Audiology Research","doi":"10.3390/audiolres16040109","source":"doaj","topics":["speech-audio"],"keywords":["speech recognition","virtual acoustics","listener position","Otorhinolaryngology"],"license":"CC0-1.0","has_fulltext":false,"has_references":false,"rights_class":"metadata_only","access":"paid","version":1,"content_endpoint":"/api/document/doi-01550a7b18e06d1fdb29","fulltext_endpoint":null,"references_endpoint":null},{"id":"doi-578f70a78beb701a4681","title":"Mix-MaxETTS: A text-to-emotional speech synthesis model based on a deep encoder–decoder structure for the transfer of secondary emotions","year":2026,"authors":["Seyyed Mahdi Hassani","Mohammad Reza Kangavari"],"author_count":2,"journal":"ETRI Journal","doi":"10.4218/etrij.2025-0058","source":"doaj","topics":["speech-audio"],"keywords":["emotion","emotional text-to-speech synthesis","encoder–decoder","speech emotion recognition","Telecommunication","Electronics"],"license":"CC0-1.0","has_fulltext":false,"has_references":false,"rights_class":"metadata_only","access":"paid","version":1,"content_endpoint":"/api/document/doi-578f70a78beb701a4681","fulltext_endpoint":null,"references_endpoint":null},{"id":"doi-38a47dc5b1ac06ccab2d","title":"Voice Cloning: A Survey of Zero-Shot and Controllable Speech Synthesis","year":2026,"authors":["Niraj Kumar Tiwari","Deepak Kumar","Asif Ekbal"],"author_count":3,"journal":"IEEE Access","doi":"10.1109/access.2026.3722351","source":"doaj","topics":["speech-audio"],"keywords":["Zero-shot voice cloning","controllable speech synthesis","neural audio codecs","codec language models","multilingual text-to-speech","streaming speech synthesis","Electrical engineering. Electronics. Nuclear engineering"],"license":"CC0-1.0","has_fulltext":false,"has_references":false,"rights_class":"metadata_only","access":"paid","version":1,"content_endpoint":"/api/document/doi-38a47dc5b1ac06ccab2d","fulltext_endpoint":null,"references_endpoint":null},{"id":"doi-6753b45391bf2fce55ed","title":"AMGAN: Adversarial Distribution Alignment of Acoustic Posteriors for Noise-Robust Automatic Speech Recognition","year":2026,"authors":["Wirya Fathy","Hadi Veisi"],"author_count":2,"journal":"IEEE Access","doi":"10.1109/access.2026.3713469","source":"doaj","topics":["speech-audio"],"keywords":["Acoustic model","automatic speech recognition","generative adversarial network","robustness","Electrical engineering. Electronics. Nuclear engineering"],"license":"CC0-1.0","has_fulltext":false,"has_references":false,"rights_class":"metadata_only","access":"paid","version":1,"content_endpoint":"/api/document/doi-6753b45391bf2fce55ed","fulltext_endpoint":null,"references_endpoint":null},{"id":"doi-2192224fa0359e6c865c","title":"LoRA-MoE Fine-Tuning for Improved Speech Recognition in People With Parkinson’s Disease","year":2026,"authors":["Seojin Yoon","Ryul Kim","Sangmin Lee"],"author_count":3,"journal":"IEEE Access","doi":"10.1109/access.2026.3722732","source":"doaj","topics":["speech-audio"],"keywords":["Dysarthric speech recognition","low-rank adaptation","mixture of experts","parameter-efficient fine-tuning","Parkinson’s disease","whisper","Electrical engineering. Electronics. Nuclear engineering"],"license":"CC0-1.0","has_fulltext":false,"has_references":false,"rights_class":"metadata_only","access":"paid","version":1,"content_endpoint":"/api/document/doi-2192224fa0359e6c865c","fulltext_endpoint":null,"references_endpoint":null},{"id":"doi-707ba15b49b6db819405","title":"Optimization of music teaching effects based on AI audio processing technology","year":2026,"authors":["Jinxu He"],"author_count":1,"journal":"Discover Artificial Intelligence","doi":"10.1007/s44163-026-01235-x","source":"doaj","topics":["speech-audio"],"keywords":["AI audio processing","Music education","Real-time feedback","Pitch analysis","Rhythm precision","Computational linguistics. Natural language processing","Electronic computers. Computer science"],"license":"CC0-1.0","has_fulltext":false,"has_references":false,"rights_class":"metadata_only","access":"paid","version":1,"content_endpoint":"/api/document/doi-707ba15b49b6db819405","fulltext_endpoint":null,"references_endpoint":null},{"id":"doi-c3563dd928dff33f7547","title":"Neural Network Based on Convolutional, Recurrent Layers and an Attention Mechanism for Visual Speech Recognition","year":2026,"authors":["D. Makar","M. Vashkevich"],"author_count":2,"journal":"Доклады Белорусского государственного университета информатики и радиоэлектроники","doi":"10.35596/1729-7648-2026-24-1-75-82","source":"doaj","topics":["speech-audio"],"keywords":["visual speech recognition","avletters2","convolutional neural network","recurrent neural network","attention mechanism","Electronics"],"license":"CC0-1.0","has_fulltext":false,"has_references":false,"rights_class":"metadata_only","access":"paid","version":1,"content_endpoint":"/api/document/doi-c3563dd928dff33f7547","fulltext_endpoint":null,"references_endpoint":null},{"id":"doi-df518bf6179dd9878b64","title":"DA-ICL: Distribution-Aware In-Context Learning for Arabic Automatic Speech Recognition Error Correction","year":2026,"authors":["Rima Sbih","Assef Jafar","Ali Kazem"],"author_count":3,"journal":"IEEE Access","doi":"10.1109/access.2026.3711662","source":"doaj","topics":["speech-audio"],"keywords":["Arabic speech recognition","large language models","HuBERT","in-context learning","distribution-aware in-context learning","parameter-efficient fine-tuning","Electrical engineering. Electronics. Nuclear engineering"],"license":"CC0-1.0","has_fulltext":false,"has_references":false,"rights_class":"metadata_only","access":"paid","version":1,"content_endpoint":"/api/document/doi-df518bf6179dd9878b64","fulltext_endpoint":null,"references_endpoint":null},{"id":"doi-9d5e2c9c3efd67943c46","title":"KANWhisper: leveraging learnable activation functions for interpretable and efficient arabic automatic speech recognition","year":2026,"authors":["Ezzaldeen Mahyoub Naji Saeed","Belal Al-sellami","Mohammed Tawfik"],"author_count":4,"journal":"Scientific Reports","doi":"10.1038/s41598-026-55863-5","source":"doaj","topics":["speech-audio"],"keywords":["Kolmogorov-Arnold networks","Arabic speech recognition","Whisper","Learnable activation functions","Interpretable AI","Transfer learning","Medicine","Science"],"license":"CC0-1.0","has_fulltext":false,"has_references":false,"rights_class":"metadata_only","access":"paid","version":1,"content_endpoint":"/api/document/doi-9d5e2c9c3efd67943c46","fulltext_endpoint":null,"references_endpoint":null},{"id":"doi-a706e663cfe69f9f576b","title":"Dictionary-Augmented Large Language Model Postprocessing for Bilingual Code-Switched Medical Speech Recognition: Development and Evaluation Study","year":2026,"authors":["Chanryeong Oh","Yul Hwangbo","Wonjoong Cheon"],"author_count":4,"journal":"Journal of Medical Internet Research","doi":"10.2196/91696","source":"doaj","topics":["speech-audio"],"keywords":["Computer applications to medicine. Medical informatics","Public aspects of medicine"],"license":"CC0-1.0","has_fulltext":false,"has_references":false,"rights_class":"metadata_only","access":"paid","version":1,"content_endpoint":"/api/document/doi-a706e663cfe69f9f576b","fulltext_endpoint":null,"references_endpoint":null},{"id":"doi-b5bc0d196dc231c9e2b3","title":"Dual-Track Residual Framework for Residual Strength-Controlled Emotional Speech Synthesis","year":2026,"authors":["Youdong Ding","Yafan Geng","Wenjing Yu"],"author_count":4,"journal":"Applied Sciences","doi":"10.3390/app16136613","source":"doaj","topics":["speech-audio"],"keywords":["emotional speech synthesis","text-to-speech synthesis","flow matching","frozen backbone adaptation","residual emotion control","duration-level prosody","Technology","Engineering (General). Civil engineering (General)"],"license":"CC0-1.0","has_fulltext":false,"has_references":false,"rights_class":"metadata_only","access":"paid","version":1,"content_endpoint":"/api/document/doi-b5bc0d196dc231c9e2b3","fulltext_endpoint":null,"references_endpoint":null},{"id":"doi-00dbb74214b6e769ef3d","title":"Automatic Speech Recognition and Acoustic Analysis for Dysarthria Assessment in Telerehabilitation: User-Centered Design and Usability Study","year":2026,"authors":["Pierre Vinet","Pierre Dillenbourg","Amelieke Slot"],"author_count":10,"journal":"JMIR Formative Research","doi":"10.2196/85230","source":"doaj","topics":["speech-audio"],"keywords":["Medicine"],"license":"CC0-1.0","has_fulltext":false,"has_references":false,"rights_class":"metadata_only","access":"paid","version":1,"content_endpoint":"/api/document/doi-00dbb74214b6e769ef3d","fulltext_endpoint":null,"references_endpoint":null},{"id":"doi-4bdc85596757b259c925","title":"A learnable, pinyin-aware post-ASR correction method with accent routing for domain-specific Chinese speech recognition","year":2026,"authors":["Yuhao Yan","Junyi Hua","Hewen Qin"],"author_count":6,"journal":"PeerJ Computer Science","doi":"10.7717/peerj-cs.4006","source":"doaj","topics":["speech-audio"],"keywords":["Mandarin ASR","Accent routing","Post-ASR correction","Pinyin-aware distance","Domain adaptation","Electronic computers. Computer science"],"license":"CC0-1.0","has_fulltext":false,"has_references":false,"rights_class":"metadata_only","access":"paid","version":1,"content_endpoint":"/api/document/doi-4bdc85596757b259c925","fulltext_endpoint":null,"references_endpoint":null},{"id":"doi-8e90c15d690753237b7b","title":"ADAPTIVE SPECTRAL NOISE REDUCTION TO IMPROVE SPEECH RECOGNITION","year":2026,"authors":["Oleh Osadchuk","Igor Olenych"],"author_count":2,"journal":"Електроніка та інформаційні технології","doi":"10.30970/eli.34.11","source":"doaj","topics":["speech-audio"],"keywords":["whisper models","background noise","noise reduction","wiener filter","spectral masking","iot","Cybernetics","Electronic computers. Computer science"],"license":"CC0-1.0","has_fulltext":false,"has_references":false,"rights_class":"metadata_only","access":"paid","version":1,"content_endpoint":"/api/document/doi-8e90c15d690753237b7b","fulltext_endpoint":null,"references_endpoint":null},{"id":"doi-1ee3be6769b575eea251","title":"Pengembangan Model Prediksi Speech Recognition dengan Algoritma Deep Learning Convolutional Neural Network","year":2026,"authors":["Abdul Halim Anshor","Aswan Supriyadi Sunge"],"author_count":2,"journal":"Jurnal Saintekom","doi":"10.33020/saintekom.v16i1.1020","source":"doaj","topics":["speech-audio"],"keywords":["sundanese language","automatic speech recognition","convolutional neural network","mel-frequency cepstral coefficients","language dialect","Technology (General)","Computer engineering. Computer hardware"],"license":"CC0-1.0","has_fulltext":false,"has_references":false,"rights_class":"metadata_only","access":"paid","version":1,"content_endpoint":"/api/document/doi-1ee3be6769b575eea251","fulltext_endpoint":null,"references_endpoint":null},{"id":"doi-1cded2e0809495b58c6e","title":"Pengembangan Asisten Pembelajaran Bahasa Inggris Portabel Berbasis Artificial Intelligence dengan Integrasi Speech Recognition dan Speech Synthesis","year":2026,"authors":["Muhammad Ricky Rizaldi","Ayu Novia Lisdawati"],"author_count":2,"journal":"Rekayasa","doi":"10.21107/rekayasa.v19i1.33686","source":"doaj","topics":["speech-audio"],"keywords":["interaksi berbasis suara","model bahasa","perangkat edge","perangkat edukatif","prototipe","sistem tertanam","Technology"],"license":"CC0-1.0","has_fulltext":false,"has_references":false,"rights_class":"metadata_only","access":"paid","version":1,"content_endpoint":"/api/document/doi-1cded2e0809495b58c6e","fulltext_endpoint":null,"references_endpoint":null},{"id":"doi-dff4f0fc4834cab614ab","title":"VQI: A Voice Quality Index for Predicting Speaker Recognition Performance","year":2026,"authors":["Ajan Ahmed","Stephanie Schuckers","Masudul H. Imtiaz"],"author_count":3,"journal":"IEEE Access","doi":"10.1109/access.2026.3703023","source":"doaj","topics":["speech-audio"],"keywords":["Biometric sample quality","quality metric","random forest","speaker recognition","speaker verification","voice quality","Electrical engineering. Electronics. Nuclear engineering"],"license":"CC0-1.0","has_fulltext":false,"has_references":false,"rights_class":"metadata_only","access":"paid","version":1,"content_endpoint":"/api/document/doi-dff4f0fc4834cab614ab","fulltext_endpoint":null,"references_endpoint":null},{"id":"doi-f54eda2d6cc14bac769c","title":"Context-Oriented Method for Resolving Lexical Ambiguities in Speech Synthesis for a Low-Resource Language","year":2026,"authors":["Elisa Izrailova","Andrey Ronzhin","Salaudin Umarkhadzhiev"],"author_count":6,"journal":"Big Data and Cognitive Computing","doi":"10.3390/bdcc10060181","source":"doaj","topics":["speech-audio"],"keywords":["word sense disambiguation","homonym","polysemantic words","low-resource languages","Chechen language","speech synthesis","Technology"],"license":"CC0-1.0","has_fulltext":false,"has_references":false,"rights_class":"metadata_only","access":"paid","version":1,"content_endpoint":"/api/document/doi-f54eda2d6cc14bac769c","fulltext_endpoint":null,"references_endpoint":null},{"id":"doi-c1d7d2359bb8cb12b682","title":"A Machine Learning Approach to Voice-Based Parkinson Disease Screening Using Multiview Spectrogram and Speech Recognition Features: Diagnostic Study","year":2026,"authors":["Arifa Zahir","Jaehong Yu","Jin-Sun Jun"],"author_count":6,"journal":"JMIR Medical Informatics","doi":"10.2196/94063","source":"doaj","topics":["speech-audio"],"keywords":["Computer applications to medicine. Medical informatics"],"license":"CC0-1.0","has_fulltext":false,"has_references":false,"rights_class":"metadata_only","access":"paid","version":1,"content_endpoint":"/api/document/doi-c1d7d2359bb8cb12b682","fulltext_endpoint":null,"references_endpoint":null},{"id":"doi-c46f9190d30600283f61","title":"Short‐ Versus Long‐Term Hearing Preservation and Speech Recognition Outcomes in Precurved and Straight Electrode Arrays","year":2026,"authors":["Lily V. Den Hartog","Andrew W. Liu","Divya A. Chari"],"author_count":3,"journal":"OTO Open","doi":"10.1002/oto2.70254","source":"doaj","topics":["speech-audio"],"keywords":["cochlear implant","electrode design","hearing preservation","precurved electrode","speech perception","straight electrode","Otorhinolaryngology","Surgery"],"license":"CC0-1.0","has_fulltext":false,"has_references":false,"rights_class":"metadata_only","access":"paid","version":1,"content_endpoint":"/api/document/doi-c46f9190d30600283f61","fulltext_endpoint":null,"references_endpoint":null},{"id":"doi-b40982f6f0d859b05aa8","title":"Automatic speech recognition for Telugu: a comparative analysis of Wav2Vec 2.0 model variants and hyperparameter tuning","year":2026,"authors":["Anvita Manne","Nikhita James","Ishaan Jain"],"author_count":4,"journal":"Frontiers in Artificial Intelligence","doi":"10.3389/frai.2026.1813668","source":"doaj","topics":["speech-audio"],"keywords":["automatic speech recognition","fine-tuning","hyperparameters","low-resource languages","Telugu","Wav2Vec 2.0","Electronic computers. Computer science"],"license":"CC0-1.0","has_fulltext":false,"has_references":false,"rights_class":"metadata_only","access":"paid","version":1,"content_endpoint":"/api/document/doi-b40982f6f0d859b05aa8","fulltext_endpoint":null,"references_endpoint":null},{"id":"doi-64ca7e819e685e8f6a07","title":"Adaptive Phoneme State Learning Architecture for Enhanced Speech Recognition Using Backpropagation Neural Network and Hidden Markov Model [version 2; peer review: 2 approved, 1 not approved]","year":2026,"authors":["Kalpana P","Priya Stella Mary I","Shivanand Gornale"],"author_count":9,"journal":"F1000Research","doi":"10.12688/f1000research.177414.2","source":"doaj","topics":["speech-audio"],"keywords":["acoustic modeling"," back propagation neural networks"," hidden markov model"," speech recognition"," voice activity detection","eng","Medicine","Science"],"license":"CC0-1.0","has_fulltext":false,"has_references":false,"rights_class":"metadata_only","access":"paid","version":1,"content_endpoint":"/api/document/doi-64ca7e819e685e8f6a07","fulltext_endpoint":null,"references_endpoint":null},{"id":"doi-937c7b8773645037cfd2","title":"Voice-Controlled Robotic System Using Arabic Speech Recognition for People with Special Needs","year":2026,"authors":["Adil Bakri","Said Fahem"],"author_count":2,"journal":"قضايا لغوية","doi":"10.61850/lij.v7i1.179","source":"doaj","topics":["speech-audio"],"keywords":["Arabic speech recognition; voice-controlled robot; assistive technology; Arduino; automatic speech recognition (ASR); special needs; MFCC.","Language. Linguistic theory. Comparative grammar"],"license":"CC0-1.0","has_fulltext":false,"has_references":false,"rights_class":"metadata_only","access":"paid","version":1,"content_endpoint":"/api/document/doi-937c7b8773645037cfd2","fulltext_endpoint":null,"references_endpoint":null},{"id":"doi-e282836e8b2630bb10b2","title":"Design of auxiliary control system for rail transit based on speech recognition","year":2026,"authors":["Wang Lujun","Jiang Bo","Ye Jingyan"],"author_count":4,"journal":"Dianzi Jishu Yingyong","doi":"10.16157/j.issn.0258-7998.257622","source":"doaj","topics":["speech-audio"],"keywords":["speech recognition","rail transit","system design","auxiliary control","Electronics"],"license":"CC0-1.0","has_fulltext":false,"has_references":false,"rights_class":"metadata_only","access":"paid","version":1,"content_endpoint":"/api/document/doi-e282836e8b2630bb10b2","fulltext_endpoint":null,"references_endpoint":null},{"id":"doi-a0ad5e6abbe29f06306e","title":"Inter‐Model Feature Fusion for Robust Low‐Resource Speech Recognition","year":2026,"authors":["Ussen Kimanuka","Ciira wa Maina","Osman Büyük"],"author_count":3,"journal":"Applied AI Letters","doi":"10.1002/ail2.70023","source":"doaj","topics":["speech-audio"],"keywords":["automatic speech recognition","co‐attentional feature‐level fusion","ensemble learning","low‐resource settings","self‐supervised learning","speech foundation model","Electronic computers. Computer science"],"license":"CC0-1.0","has_fulltext":false,"has_references":false,"rights_class":"metadata_only","access":"paid","version":1,"content_endpoint":"/api/document/doi-a0ad5e6abbe29f06306e","fulltext_endpoint":null,"references_endpoint":null},{"id":"doi-ea7fe9ab3535d7b7a47b","title":"Voice-Based Structured Nursing Documentation Using Automatic Speech Recognition and Large Language Models: Development and Evaluation Study","year":2026,"authors":["Meng-Han Su","Wei-Chun Wang","Yi-Min Hsu"],"author_count":6,"journal":"JMIR Nursing","doi":"10.2196/88567","source":"doaj","topics":["speech-audio"],"keywords":["Nursing"],"license":"CC0-1.0","has_fulltext":false,"has_references":false,"rights_class":"metadata_only","access":"paid","version":1,"content_endpoint":"/api/document/doi-ea7fe9ab3535d7b7a47b","fulltext_endpoint":null,"references_endpoint":null},{"id":"doi-9f972adebdf75e55806c","title":"Automatic speech recognition and dialect identification for Arabic and Saudi dialects: a systematic literature review","year":2026,"authors":["Alanoud S. A. Alhaissoni","Wael M. S. Yafooz","Abdullah Alsaeedi"],"author_count":3,"journal":"Journal of King Saud University: Computer and Information Sciences","doi":"10.1007/s44443-026-00609-6","source":"doaj","topics":["speech-audio"],"keywords":["Arabic dialect","Automatic speech recognition (ASR)","Dialect identification (DID)","Deep learning (DL)","Machine learning (ML)","Saudi Arabic","Electronic computers. Computer science"],"license":"CC0-1.0","has_fulltext":false,"has_references":false,"rights_class":"metadata_only","access":"paid","version":1,"content_endpoint":"/api/document/doi-9f972adebdf75e55806c","fulltext_endpoint":null,"references_endpoint":null},{"id":"doi-dfaefd606d49be0ca79e","title":"Enhancing multilingual automatic speech recognition for low-resource code-switched languages: a scalable data augmentation strategy","year":2026,"authors":["Mohab Mostafa Morsi","Radwa Fathalla","Sherif Abdou"],"author_count":4,"journal":"PeerJ Computer Science","doi":"10.7717/peerj-cs.3877","source":"doaj","topics":["speech-audio"],"keywords":["Arabic-English","Egyptian colloquial Arabic","Code-switching","Data augmentation","Automatic speech recognition","Text-to-speech","Electronic computers. Computer science"],"license":"CC0-1.0","has_fulltext":false,"has_references":false,"rights_class":"metadata_only","access":"paid","version":1,"content_endpoint":"/api/document/doi-dfaefd606d49be0ca79e","fulltext_endpoint":null,"references_endpoint":null},{"id":"doi-236f9a85d9200989f168","title":"Speech Recognition–Based Dietary Assessment Tool for Older Adults: Validation and Usability Study","year":2026,"authors":["Yoonjee ‍Sung","Soyoung Jung","Hae Jin Kang"],"author_count":8,"journal":"JMIR Aging","doi":"10.2196/81336","source":"doaj","topics":["speech-audio"],"keywords":["Geriatrics"],"license":"CC0-1.0","has_fulltext":false,"has_references":false,"rights_class":"metadata_only","access":"paid","version":1,"content_endpoint":"/api/document/doi-236f9a85d9200989f168","fulltext_endpoint":null,"references_endpoint":null},{"id":"doi-feeec7eda1bbba8d0157","title":"Case Report: Tailored automatic speech recognition in global aphasia with dysarthria - a single case proof of concept","year":2026,"authors":["Davide Mulfari","Davide Cardile","Davide Cardile"],"author_count":10,"journal":"Frontiers in Rehabilitation Sciences","doi":"10.3389/fresc.2026.1813312","source":"doaj","topics":["speech-audio"],"keywords":["aphasia","assistive technology","automatic speech recognition","dysarthria","rehabilitation","speaker-dependent ASR","Other systems of medicine","Medical technology"],"license":"CC0-1.0","has_fulltext":false,"has_references":false,"rights_class":"metadata_only","access":"paid","version":1,"content_endpoint":"/api/document/doi-feeec7eda1bbba8d0157","fulltext_endpoint":null,"references_endpoint":null},{"id":"doi-c8d285de7c8f3cb10923","title":"Research on hydrodynamic simulation and speech recognition method based on PLDA model","year":2026,"authors":["Wu Lixian","Lai Weiwei","Bu li"],"author_count":6,"journal":"Journal of Applied Science and Engineering","doi":"10.6180/jase.202609_32.057","source":"doaj","topics":["speech-audio"],"keywords":["plda model","speech recognition","hydrodynamic simulation","cross-domain migration","Engineering (General). Civil engineering (General)","Chemical engineering","Physics"],"license":"CC0-1.0","has_fulltext":false,"has_references":false,"rights_class":"metadata_only","access":"paid","version":1,"content_endpoint":"/api/document/doi-c8d285de7c8f3cb10923","fulltext_endpoint":null,"references_endpoint":null},{"id":"doi-f973bdfc8d345b00cc9b","title":"JODAL: Joint Domain Adversarial Learning for TTS-Augmented Automatic Speech Recognition","year":2026,"authors":["June-Woo Kim","Ho-Young Jung"],"author_count":2,"journal":"Mathematics","doi":"10.3390/math14101669","source":"doaj","topics":["speech-audio"],"keywords":["automatic speech recognition","text-to-speech","domain adaptation","data augmentation","synthetic data","deep learning applications","Mathematics"],"license":"CC0-1.0","has_fulltext":false,"has_references":false,"rights_class":"metadata_only","access":"paid","version":1,"content_endpoint":"/api/document/doi-f973bdfc8d345b00cc9b","fulltext_endpoint":null,"references_endpoint":null},{"id":"doi-0766b986afaa9bb9ed67","title":"Speech Recognition with an fMRISNN Constrained by Human Functional Brain Networks: A Study of Enhanced MFCC-Driven Sparse Spike Encoding","year":2026,"authors":["Lei Guo","Nancheng Ma","Zhuoxuan Wang"],"author_count":4,"journal":"Biomimetics","doi":"10.3390/biomimetics11050302","source":"doaj","topics":["speech-audio"],"keywords":["SNN","fMRI","functional brain network","sparse spike encoding","speech recognition","Technology"],"license":"CC0-1.0","has_fulltext":false,"has_references":false,"rights_class":"metadata_only","access":"paid","version":1,"content_endpoint":"/api/document/doi-0766b986afaa9bb9ed67","fulltext_endpoint":null,"references_endpoint":null},{"id":"doi-c28f17ffe566d8bfb689","title":"PRL-DAS: Robust Heliox Speech Recognition for Unaligned Low-Resource Data","year":2026,"authors":["Yonghong Chen","Guoqi Zhang","Wanzhi Wen"],"author_count":4,"journal":"Big Data and Cognitive Computing","doi":"10.3390/bdcc10050157","source":"doaj","topics":["speech-audio"],"keywords":["heliox speech","robust speech recognition","Whisper","LoRA","duration-adaptive speed","low-resource speech recognition","Technology"],"license":"CC0-1.0","has_fulltext":false,"has_references":false,"rights_class":"metadata_only","access":"paid","version":1,"content_endpoint":"/api/document/doi-c28f17ffe566d8bfb689","fulltext_endpoint":null,"references_endpoint":null},{"id":"doi-31fb4bad70f5c5924b1d","title":"Systematic Annotation Framework for Robust Speech Recognition","year":2026,"authors":["Zhong Wang","Chunjie Cao","Xia Xie"],"author_count":5,"journal":"Applied Sciences","doi":"10.3390/app16104850","source":"doaj","topics":["speech-audio"],"keywords":["automatic speech recognition","robust speech recognition","speech annotation","dialect speech","corpus construction","quality control","Technology","Engineering (General). Civil engineering (General)"],"license":"CC0-1.0","has_fulltext":false,"has_references":false,"rights_class":"metadata_only","access":"paid","version":1,"content_endpoint":"/api/document/doi-31fb4bad70f5c5924b1d","fulltext_endpoint":null,"references_endpoint":null},{"id":"doi-c455e354837c4fdbe9d7","title":"Advancing Speech Therapy with AI: A Comparative Analysis of Speech Recognition Models","year":2026,"authors":["Sameh Zarif","Sama Ahmed","Marian Wagdy"],"author_count":3,"journal":"Journal of Computing and Communication","doi":"10.21608/jocc.2026.411864.1102","source":"doaj","topics":["speech-audio"],"keywords":["Speech Therapy","Speech recognition","AI-based Pronunciation","Deep learning","ASR","Electronic computers. Computer science","Communication. Mass media"],"license":"CC0-1.0","has_fulltext":false,"has_references":false,"rights_class":"metadata_only","access":"paid","version":1,"content_endpoint":"/api/document/doi-c455e354837c4fdbe9d7","fulltext_endpoint":null,"references_endpoint":null},{"id":"doi-eae7847b76533062c1a9","title":"Noise-Resilient Visual Speech Recognition for Intention Decoding in Human–Robot Collaboration","year":2026,"authors":["Haoze Zhou","Lifeng Qin","Dan Zhao"],"author_count":3,"journal":"IEEE Access","doi":"10.1109/access.2026.3690950","source":"doaj","topics":["speech-audio"],"keywords":["Human–robot interaction","human–robot collaboration","visual speech recognition","multimodalities fusion","transfer learning","assembly","Electrical engineering. Electronics. Nuclear engineering"],"license":"CC0-1.0","has_fulltext":false,"has_references":false,"rights_class":"metadata_only","access":"paid","version":1,"content_endpoint":"/api/document/doi-eae7847b76533062c1a9","fulltext_endpoint":null,"references_endpoint":null},{"id":"doi-170484e24f8b1ca40f6b","title":"Bidirectional Kazakh Sign Language prosody-aware translation using computer vision and speech recognition techniques","year":2026,"authors":["Mukhtar Zhassuzak","Mukhtar Zhassuzak","Zholdas Buribayev"],"author_count":8,"journal":"Frontiers in Artificial Intelligence","doi":"10.3389/frai.2026.1835419","source":"doaj","topics":["speech-audio"],"keywords":["human-computer interaction","Liquid Neural Networks","prosody prediction","sign language translation","speech synthesis","Electronic computers. Computer science"],"license":"CC0-1.0","has_fulltext":false,"has_references":false,"rights_class":"metadata_only","access":"paid","version":1,"content_endpoint":"/api/document/doi-170484e24f8b1ca40f6b","fulltext_endpoint":null,"references_endpoint":null},{"id":"doi-836fea44da4b92c1f7fb","title":"Automatic Speech Recognition and Large Language Models for Multilingual Pathology Report Generation: Proof-of-Concept Study","year":2026,"authors":["Kuan-Hsun Lin","Chia-Ping Chang","Chen-Tsung Kuo"],"author_count":9,"journal":"JMIR Formative Research","doi":"10.2196/90814","source":"doaj","topics":["speech-audio"],"keywords":["Medicine"],"license":"CC0-1.0","has_fulltext":false,"has_references":false,"rights_class":"metadata_only","access":"paid","version":1,"content_endpoint":"/api/document/doi-836fea44da4b92c1f7fb","fulltext_endpoint":null,"references_endpoint":null},{"id":"doi-a0c093bf7a6ca2082817","title":"LoRA-enhanced whisper for resource-efficient heliox speech recognition","year":2026,"authors":["Weichang Mao","Haojie Gu","Jia He"],"author_count":5,"journal":"Scientific Reports","doi":"10.1038/s41598-026-38201-7","source":"doaj","topics":["speech-audio"],"keywords":["Helium speech recognition","LoRA","Whisper","Hotword bias","LM reranking","Medicine","Science"],"license":"CC0-1.0","has_fulltext":false,"has_references":false,"rights_class":"metadata_only","access":"paid","version":1,"content_endpoint":"/api/document/doi-a0c093bf7a6ca2082817","fulltext_endpoint":null,"references_endpoint":null},{"id":"doi-a7f51b6a4612cdf7fb4f","title":"Prosody Disruptor for Voice Protection Against Unauthorized Speech Synthesis","year":2026,"authors":["Seoyoung Park","An Thien Nguyen","Thien-Phuc Doan"],"author_count":4,"journal":"IEEE Access","doi":"10.1109/access.2026.3688217","source":"doaj","topics":["speech-audio"],"keywords":["Adversarial machine learning","biometric authentication","deep learning","generative AI","information security","machine learning","Electrical engineering. Electronics. Nuclear engineering"],"license":"CC0-1.0","has_fulltext":false,"has_references":false,"rights_class":"metadata_only","access":"paid","version":1,"content_endpoint":"/api/document/doi-a7f51b6a4612cdf7fb4f","fulltext_endpoint":null,"references_endpoint":null},{"id":"doi-59303301574dd7c863cd","title":"Retraction: Application of an Isolated Word Speech Recognition System in the Field of Mental Health Consultation: Development and Usability Study","year":2026,"authors":[],"author_count":0,"journal":"JMIR Medical Informatics","doi":"10.2196/98661","source":"doaj","topics":["speech-audio"],"keywords":["Computer applications to medicine. Medical informatics"],"license":"CC0-1.0","has_fulltext":false,"has_references":false,"rights_class":"metadata_only","access":"paid","version":1,"content_endpoint":"/api/document/doi-59303301574dd7c863cd","fulltext_endpoint":null,"references_endpoint":null},{"id":"doi-7158ab869903cfce6cbe","title":"Automatic Speech Recognition for Javanese Language using Wav2Vec 2.0 with Finetuning","year":2026,"authors":["Johanes Setiawan","Ardytha Luthfiarta","Adhitya Nugraha "],"author_count":6,"journal":"Jurnal Teknologi dan Sistem Informasi","doi":"10.25077/teknosi.v12i1.2026.1-9","source":"doaj","topics":["speech-audio"],"keywords":["Javanese Language","Wav2Vec 2.0","Speech to Text","Deep Learning","Finetuning","Electronic computers. Computer science"],"license":"CC0-1.0","has_fulltext":false,"has_references":false,"rights_class":"metadata_only","access":"paid","version":1,"content_endpoint":"/api/document/doi-7158ab869903cfce6cbe","fulltext_endpoint":null,"references_endpoint":null},{"id":"doi-78e0afc15ad0b7ff8c84","title":"Speech recognition performance with dual-microphone audio processors in mandarin-speaking cochlear implant users","year":2026,"authors":["Kailong Yin","Shuo Chen","Fei Wang"],"author_count":4,"journal":"Frontiers in Neuroscience","doi":"10.3389/fnins.2026.1767325","source":"doaj","topics":["speech-audio"],"keywords":["adaptive noise reduction","audio processors","cochlear implants","mandarin Chinese users","speech recognition in noise","Neurosciences. Biological psychiatry. Neuropsychiatry"],"license":"CC0-1.0","has_fulltext":false,"has_references":false,"rights_class":"metadata_only","access":"paid","version":1,"content_endpoint":"/api/document/doi-78e0afc15ad0b7ff8c84","fulltext_endpoint":null,"references_endpoint":null},{"id":"doi-d7d18ad1edd7cd72f52f","title":"Deep Hidden Semi-Markov Model-Based Speech Synthesis","year":2026,"authors":["Yoshihiko Nankaku","Takato Fujimoto","Takenori Yoshimura"],"author_count":7,"journal":"IEEE Access","doi":"10.1109/access.2026.3683761","source":"doaj","topics":["speech-audio"],"keywords":["Speech synthesis","deep neural networks","sequence-to-sequence models","attention mechanisms","hidden semi-Markov models","variational autoencoders","Electrical engineering. Electronics. Nuclear engineering"],"license":"CC0-1.0","has_fulltext":false,"has_references":false,"rights_class":"metadata_only","access":"paid","version":1,"content_endpoint":"/api/document/doi-d7d18ad1edd7cd72f52f","fulltext_endpoint":null,"references_endpoint":null},{"id":"doi-fad4ec94b7e678cf28e8","title":"Visual Implicit Learning and Speech Recognition in Adult Post-Lingual Cochlear Implant Users","year":2026,"authors":["Ranin Khayr","Riyad Khnifes","Karen Banai"],"author_count":3,"journal":"Trends in Hearing","doi":"10.1177/23312165261434604","source":"doaj","topics":["speech-audio"],"keywords":["Otorhinolaryngology"],"license":"CC0-1.0","has_fulltext":false,"has_references":false,"rights_class":"metadata_only","access":"paid","version":1,"content_endpoint":"/api/document/doi-fad4ec94b7e678cf28e8","fulltext_endpoint":null,"references_endpoint":null},{"id":"doi-e8e1267f4c8b34b98947","title":"Research Progress in Target Audio Processing Methods Based on Pre⁃trained Models","year":2026,"authors":["LIU Ju","MA Hao","LI Xiaohang"],"author_count":8,"journal":"Shuju Caiji Yu Chuli","doi":"10.16337/j.1004⁃9037.2026.02.007","source":"doaj","topics":["speech-audio"],"keywords":["target audio processing","pre-trained model","parameter-efficient fine-tuning","target sound extraction","target speaker speech recognition","contrastive learning","Information technology","Technology"],"license":"CC0-1.0","has_fulltext":false,"has_references":false,"rights_class":"metadata_only","access":"paid","version":1,"content_endpoint":"/api/document/doi-e8e1267f4c8b34b98947","fulltext_endpoint":null,"references_endpoint":null},{"id":"doi-ac4e78ca6ceaa35f15ea","title":"Universal Adversarial Example Generation Method with High Transferability for Transformer‑Based Speech Recognition Models","year":2026,"authors":["WANG Zhen","HAN Jiqing","HE Yongjun"],"author_count":5,"journal":"Shuju Caiji Yu Chuli","doi":"10.16337/j.1004⁃9037.2026.01.007","source":"doaj","topics":["speech-audio"],"keywords":["speech recognition","adversarial examples","black-box attack","attention mechanism","Information technology","Technology"],"license":"CC0-1.0","has_fulltext":false,"has_references":false,"rights_class":"metadata_only","access":"paid","version":1,"content_endpoint":"/api/document/doi-ac4e78ca6ceaa35f15ea","fulltext_endpoint":null,"references_endpoint":null},{"id":"doi-ec2cea3b84162d3cf686","title":"Multilingual Self-Supervised Fine-Tuning for Low-Resource Telugu Automatic Speech Recognition","year":2026,"authors":["Srivarthin Vaddepalli","Renjith Prabhavathi Neelakandan"],"author_count":2,"journal":"IEEE Access","doi":"10.1109/access.2026.3678805","source":"doaj","topics":["speech-audio"],"keywords":["Automatic speech recognition","Telugu language","low-resource ASR","transfer learning","Wav2Vec2-XLSR-53","KenLM","Electrical engineering. Electronics. Nuclear engineering"],"license":"CC0-1.0","has_fulltext":false,"has_references":false,"rights_class":"metadata_only","access":"paid","version":1,"content_endpoint":"/api/document/doi-ec2cea3b84162d3cf686","fulltext_endpoint":null,"references_endpoint":null},{"id":"doi-35a81b9d9ab88dec29cd","title":"Toward Unified Chinese Multi-Dialectal Speech Recognition via Pinyin Intermediate Representation","year":2026,"authors":["Junjie Cheng","Xiaorong Wu"],"author_count":2,"journal":"IEEE Access","doi":"10.1109/access.2026.3681804","source":"doaj","topics":["speech-audio"],"keywords":["Multi-dialect speech recognition","Pinyin intermediate representation","cross-dialect phonetic sharing","unified multi-dialect framework","low-resource dialects","Electrical engineering. Electronics. Nuclear engineering"],"license":"CC0-1.0","has_fulltext":false,"has_references":false,"rights_class":"metadata_only","access":"paid","version":1,"content_endpoint":"/api/document/doi-35a81b9d9ab88dec29cd","fulltext_endpoint":null,"references_endpoint":null},{"id":"doi-bc1745d9bc88a015afd2","title":"Exploring the Use of AI-Based Speech Recognition Platforms to Support Elementary Students’ Speaking Skills in Distance Learning","year":2026,"authors":["Saddam Fathurrachman"],"author_count":1,"journal":"Inspiring","doi":"10.35905/inspiring.v9i1.16462","source":"doaj","topics":["speech-audio"],"keywords":["Distance Learning","Elementary Students","Google Assistant","Speaking Skills","Speech Recognition","Philology. Linguistics"],"license":"CC0-1.0","has_fulltext":false,"has_references":false,"rights_class":"metadata_only","access":"paid","version":1,"content_endpoint":"/api/document/doi-bc1745d9bc88a015afd2","fulltext_endpoint":null,"references_endpoint":null},{"id":"doi-030b55d3c0b716d77807","title":"Multimodal AI-enabled Feedback Mechanism for Oral Foreign Language Proficiency: Integrating Speech Recognition and Sentiment Analysis","year":2026,"authors":["Yingying He"],"author_count":1,"journal":"Journal of Applied Science and Engineering","doi":"10.6180/jase.202608_31.068","source":"doaj","topics":["speech-audio"],"keywords":["multimodal ai","oral foreign language proficiency","speech recognition","sentiment analysis","feedback mechanism","Engineering (General). Civil engineering (General)","Chemical engineering","Physics"],"license":"CC0-1.0","has_fulltext":false,"has_references":false,"rights_class":"metadata_only","access":"paid","version":1,"content_endpoint":"/api/document/doi-030b55d3c0b716d77807","fulltext_endpoint":null,"references_endpoint":null},{"id":"doi-ea07a4ac22e656ca91d2","title":"Strengthening Early Childhood English Pronunciation through Tiny Pronounce: A Speech Recognition–Based Learning Innovation","year":2026,"authors":["Ari Astutik","Anisa’ Nur Azizah","Sulistiyani Sulistiyani"],"author_count":3,"journal":"Journal of Community Service and Empowerment","doi":"10.22219/jcse.v7i1.42333","source":"doaj","topics":["speech-audio"],"keywords":["Pronunciation","Early childhood","Speech recognition","Tiny Pronounce","Interactive learning","Human settlements. Communities"],"license":"CC0-1.0","has_fulltext":false,"has_references":false,"rights_class":"metadata_only","access":"paid","version":1,"content_endpoint":"/api/document/doi-ea07a4ac22e656ca91d2","fulltext_endpoint":null,"references_endpoint":null},{"id":"doi-0150a3719dd4f298b994","title":"A CTC-Based Speech Recognition Network Fusing Local Convolution and Global Attention","year":2026,"authors":["Huijuan Hu","Chenyang Tang","Ping Tan"],"author_count":4,"journal":"Sensors","doi":"10.3390/s26061865","source":"doaj","topics":["speech-audio"],"keywords":["automatic speech recognition","wav2vec 2.0","information fusion","dual-branch architecture","task-aware gating","CTC alignment","Chemical technology"],"license":"CC0-1.0","has_fulltext":false,"has_references":false,"rights_class":"metadata_only","access":"paid","version":1,"content_endpoint":"/api/document/doi-0150a3719dd4f298b994","fulltext_endpoint":null,"references_endpoint":null},{"id":"doi-9082d5430a9ebffa8dbc","title":"Amharic Speech Recognition Based on Phoneme Context-Sensitivity Analysis and Error Diagnostics","year":2026,"authors":["Eyasu Demissie Yenieneh","Surafiel Habib Asefa","Yaregal Assabie"],"author_count":3,"journal":"IEEE Access","doi":"10.1109/access.2026.3674426","source":"doaj","topics":["speech-audio"],"keywords":["Automatic speech recognition","amharic speech recognition","context sensitivity","low-resource languages","transformer","Wav2vec","Electrical engineering. Electronics. Nuclear engineering"],"license":"CC0-1.0","has_fulltext":false,"has_references":false,"rights_class":"metadata_only","access":"paid","version":1,"content_endpoint":"/api/document/doi-9082d5430a9ebffa8dbc","fulltext_endpoint":null,"references_endpoint":null},{"id":"doi-6afe896bf9fdd5e375be","title":"Interactive web-based text-to-speech and speech recognition media for enhancing Arabic listening proficiency","year":2026,"authors":["Luthfiyatuz Zuhriyah","Asep Sunarko","Ahmad Zuhdi"],"author_count":4,"journal":"Al-Lisan: Jurnal Bahasa","doi":"10.30603/al.v11i1.7257","source":"doaj","topics":["speech-audio"],"keywords":["arabic listening;","interactive web-based;","text-to-speech;","speech recognition media","English language","Philology. Linguistics"],"license":"CC0-1.0","has_fulltext":false,"has_references":false,"rights_class":"metadata_only","access":"paid","version":1,"content_endpoint":"/api/document/doi-6afe896bf9fdd5e375be","fulltext_endpoint":null,"references_endpoint":null},{"id":"doi-21613951f8698bcf49bd","title":"A fully integrated, stretchable bioelectronic interface for sEMG-based motion and speech recognition","year":2026,"authors":["Chang Liu","Xiaoying Zhu","Senhao Zhang"],"author_count":5,"journal":"Materials & Design","doi":"10.1016/j.matdes.2026.115718","source":"doaj","topics":["speech-audio"],"keywords":["Surface electromyography","Stretchable electronics","Structural design","Machine learning","Human-machine interaction","Epidermal bioelectronics","Materials of engineering and construction. Mechanics of materials"],"license":"CC0-1.0","has_fulltext":false,"has_references":false,"rights_class":"metadata_only","access":"paid","version":1,"content_endpoint":"/api/document/doi-21613951f8698bcf49bd","fulltext_endpoint":null,"references_endpoint":null},{"id":"doi-c734c9b85ecb8e029667","title":"Quantifying the Relationship Between Speech Quality Metrics and Biometric Speaker Recognition Performance Under Acoustic Degradation","year":2026,"authors":["Ajan Ahmed","Masudul H. Imtiaz"],"author_count":2,"journal":"Signals","doi":"10.3390/signals7010007","source":"doaj","topics":["speech-audio"],"keywords":["speaker recognition","voice biometrics","speech quality assessment","PESQ","self-supervised learning","signal degradation","Applied mathematics. Quantitative methods"],"license":"CC0-1.0","has_fulltext":false,"has_references":false,"rights_class":"metadata_only","access":"paid","version":1,"content_endpoint":"/api/document/doi-c734c9b85ecb8e029667","fulltext_endpoint":null,"references_endpoint":null},{"id":"doi-f126e1ab515f709044d4","title":"A Mixed Reality Tool with Automatic Speech Recognition for 3D CAD Based Visualization and Automatic Dimension Generation in the Industry 5.0 Shipyard","year":2026,"authors":["Aida Vidal-Balea","Antón Valladares-Poncela","Javier Vilar-Martínez"],"author_count":5,"journal":"Multimodal Technologies and Interaction","doi":"10.3390/mti10020013","source":"doaj","topics":["speech-audio"],"keywords":["Extended Reality","Augmented Reality","Mixed Reality","Microsoft HoloLens","MRTK","ASR","Technology","Science"],"license":"CC0-1.0","has_fulltext":false,"has_references":false,"rights_class":"metadata_only","access":"paid","version":1,"content_endpoint":"/api/document/doi-f126e1ab515f709044d4","fulltext_endpoint":null,"references_endpoint":null},{"id":"doi-ebea7f4dcf00bc0d0af9","title":"The Effect of Modulation Enhancement Scheme on Speech Recognition in Spatial Noise Among Young Adults with Normal Hearing","year":2026,"authors":["Vibha Kanagokar","M. A. Yashu","Jayashree S. Bhat"],"author_count":4,"journal":"Audiology Research","doi":"10.3390/audiolres16010026","source":"doaj","topics":["speech-audio"],"keywords":["temporal envelope","envelope expansion","envelope enhancement","spatial release from masking","interaural time difference","interaural coherence","Otorhinolaryngology"],"license":"CC0-1.0","has_fulltext":false,"has_references":false,"rights_class":"metadata_only","access":"paid","version":1,"content_endpoint":"/api/document/doi-ebea7f4dcf00bc0d0af9","fulltext_endpoint":null,"references_endpoint":null},{"id":"doi-7ccbc4a56ad43a51e161","title":"Modern Speech Recognition for Romanian Language","year":2026,"authors":["Remus-Dan Ungureanu","Mihai Dascalu"],"author_count":2,"journal":"Applied Sciences","doi":"10.3390/app16041928","source":"doaj","topics":["speech-audio"],"keywords":["Romanian automatic speech recognition","low-resource language","wav2vec 2.0","Conformer","weakly supervised learning","Technology","Engineering (General). Civil engineering (General)","Biology (General)"],"license":"CC0-1.0","has_fulltext":false,"has_references":false,"rights_class":"metadata_only","access":"paid","version":1,"content_endpoint":"/api/document/doi-7ccbc4a56ad43a51e161","fulltext_endpoint":null,"references_endpoint":null},{"id":"doi-6bda0f5c890f939a6c58","title":"Robust Dysarthric Speech Recognition with GAN Enhancement and LLM Correction","year":2026,"authors":["Yibo He","Kah Phooi Seng","Chee Shen Lim"],"author_count":4,"journal":"Advanced Intelligent Systems","doi":"10.1002/aisy.202500873","source":"doaj","topics":["speech-audio"],"keywords":["dysarthric speech recognition","error correction","generative adversarial network enhancement","large language models","large language model","robust speech recognition","Computer engineering. Computer hardware","Control engineering systems. Automatic machinery (General)"],"license":"CC0-1.0","has_fulltext":false,"has_references":false,"rights_class":"metadata_only","access":"paid","version":1,"content_endpoint":"/api/document/doi-6bda0f5c890f939a6c58","fulltext_endpoint":null,"references_endpoint":null},{"id":"doi-03f091a2de866b65b08f","title":"Objective Evaluation of Prosody and Intelligibility in Speech Synthesis via Conditional Prediction of Discrete Tokens","year":2026,"authors":["Ismail Rasim Ulgen","Zongyang Du","Junchen Lu"],"author_count":5,"journal":"IEEE Open Journal of Signal Processing","doi":"10.1109/ojsp.2026.3653666","source":"doaj","topics":["speech-audio"],"keywords":["Speech generation","objective evaluation","intelligibility","speech tokens","prosody evaluation","Electrical engineering. Electronics. Nuclear engineering"],"license":"CC0-1.0","has_fulltext":false,"has_references":false,"rights_class":"metadata_only","access":"paid","version":1,"content_endpoint":"/api/document/doi-03f091a2de866b65b08f","fulltext_endpoint":null,"references_endpoint":null},{"id":"doi-bc8147e061c4eb59ea0d","title":"The impact of AI-driven speech recognition on listening comprehension and pronunciation accuracy in English language teaching","year":2026,"authors":["Fakher Rahim","Raikan Ysmailova Apzhaparovna"],"author_count":2,"journal":"Discover Computing","doi":"10.1007/s10791-026-09992-0","source":"doaj","topics":["speech-audio"],"keywords":["AI-assisted learning","Automatic speech recognition (ASR)","Educational technology","English language teaching","Electronic computers. Computer science"],"license":"CC0-1.0","has_fulltext":false,"has_references":false,"rights_class":"metadata_only","access":"paid","version":1,"content_endpoint":"/api/document/doi-bc8147e061c4eb59ea0d","fulltext_endpoint":null,"references_endpoint":null},{"id":"doi-5bb5e8db64c7e9f0935c","title":"An Input-synchronous Blockwise Decoding Algorithm for CTC-AED Speech Recognition","year":2026,"authors":["Iurii Lezhenin","Natalia Bogach"],"author_count":2,"journal":"Информатика и автоматизация","doi":"10.15622/ia.25.1.5","source":"doaj","topics":["speech-audio"],"keywords":["streaming automatic speech recognition (asr)","blockwise decoding","end-to-end","ctc","aed","Electronic computers. Computer science"],"license":"CC0-1.0","has_fulltext":false,"has_references":false,"rights_class":"metadata_only","access":"paid","version":1,"content_endpoint":"/api/document/doi-5bb5e8db64c7e9f0935c","fulltext_endpoint":null,"references_endpoint":null},{"id":"arxiv-2310.11379","title":"Robust Wake-Up Word Detection by Two-stage Multi-resolution Ensembles","year":2026,"authors":["Fernando López","Jordi Luque","Carlos Segura"],"author_count":4,"journal":null,"doi":null,"source":"arxiv","topics":["speech-audio"],"keywords":["cs.SD","cs.CL","eess.AS"],"license":"CC-BY-4.0","has_fulltext":false,"has_references":false,"rights_class":"licensed_content","access":"paid","version":1,"content_endpoint":"/api/document/arxiv-2310.11379","fulltext_endpoint":null,"references_endpoint":null},{"id":"arxiv-2603.22252","title":"SelfTTS: cross-speaker style transfer through explicit embedding disentanglement and self-refinement using self-augmentation","year":2026,"authors":["Lucas H. Ueda","João G. T. Lima","Pedro R. Corrêa"],"author_count":6,"journal":null,"doi":null,"source":"arxiv","topics":["speech-audio"],"keywords":["eess.AS","cs.SD"],"license":"CC0-1.0","has_fulltext":false,"has_references":false,"rights_class":"metadata_only","access":"paid","version":1,"content_endpoint":"/api/document/arxiv-2603.22252","fulltext_endpoint":null,"references_endpoint":null},{"id":"arxiv-2603.23673","title":"Crab: Multi Layer Contrastive Supervision to Improve Speech Emotion Recognition Under Both Acted and Natural Speech Condition","year":2026,"authors":["Lucas H. Ueda","João G. T. Lima","Paula D. P. Costa"],"author_count":3,"journal":null,"doi":null,"source":"arxiv","topics":["speech-audio"],"keywords":["eess.AS","cs.SD"],"license":"CC0-1.0","has_fulltext":false,"has_references":false,"rights_class":"metadata_only","access":"paid","version":1,"content_endpoint":"/api/document/arxiv-2603.23673","fulltext_endpoint":null,"references_endpoint":null},{"id":"arxiv-2609.09940","title":"NVV-Locator: From Transcript Tags to Acoustic Boundaries for Fine-Grained Nonverbal Vocalization Grounding","year":2026,"authors":["Yuang Cao","Bingshen Mu","Zhennan Lin"],"author_count":10,"journal":null,"doi":null,"source":"arxiv","topics":["speech-audio"],"keywords":["eess.AS","cs.SD"],"license":"CC0-1.0","has_fulltext":false,"has_references":false,"rights_class":"metadata_only","access":"paid","version":1,"content_endpoint":"/api/document/arxiv-2609.09940","fulltext_endpoint":null,"references_endpoint":null},{"id":"arxiv-2609.38203","title":"Automatic estimation of verbal fluency index in people with Motor Neuron Disease using ASR alignment and pause modelling","year":2026,"authors":["Bahman Mirheidari","Leslie Ing","Daniel Blackburn"],"author_count":6,"journal":null,"doi":null,"source":"arxiv","topics":["speech-audio"],"keywords":["cs.CL","cs.AI","eess.AS"],"license":"CC-BY-4.0","has_fulltext":false,"has_references":false,"rights_class":"licensed_content","access":"paid","version":1,"content_endpoint":"/api/document/arxiv-2609.38203","fulltext_endpoint":null,"references_endpoint":null},{"id":"arxiv-2609.38440","title":"Monotonicity-Guided Semantic Alignment for Zero-shot Multispeaker Image-to-Speech Synthesis","year":2026,"authors":["Lijun Wang","Yixian Lu","Shogo Okada"],"author_count":3,"journal":null,"doi":null,"source":"arxiv","topics":["speech-audio"],"keywords":["eess.AS"],"license":"CC0-1.0","has_fulltext":false,"has_references":false,"rights_class":"metadata_only","access":"paid","version":1,"content_endpoint":"/api/document/arxiv-2609.38440","fulltext_endpoint":null,"references_endpoint":null},{"id":"arxiv-2609.38501","title":"Voices as Handles: Reasoning about Speaker Identity with Frozen Text LLMs","year":2026,"authors":["Runqiu Xu","Zhisheng Zheng","David Harwath"],"author_count":3,"journal":null,"doi":null,"source":"arxiv","topics":["speech-audio"],"keywords":["eess.AS","cs.SD"],"license":"CC0-1.0","has_fulltext":false,"has_references":false,"rights_class":"metadata_only","access":"paid","version":1,"content_endpoint":"/api/document/arxiv-2609.38501","fulltext_endpoint":null,"references_endpoint":null},{"id":"arxiv-2609.38658","title":"Tacit-TTS: From Autoregressive Decoding to Masked Prediction for Efficient Transcript-Free Voice Cloning","year":2026,"authors":["Jian Chen","You Zhang","Mark Vinton"],"author_count":3,"journal":null,"doi":null,"source":"arxiv","topics":["speech-audio"],"keywords":["eess.AS","cs.CL","cs.SD"],"license":"CC0-1.0","has_fulltext":false,"has_references":false,"rights_class":"metadata_only","access":"paid","version":1,"content_endpoint":"/api/document/arxiv-2609.38658","fulltext_endpoint":null,"references_endpoint":null},{"id":"arxiv-2609.38867","title":"Talk2Agent: Benchmarking Voice Interfaces for Text Agents","year":2026,"authors":["Terumi Chiba","Guangzhi Sun","Zheqi Yuan"],"author_count":4,"journal":null,"doi":null,"source":"arxiv","topics":["speech-audio"],"keywords":["cs.AI","eess.AS"],"license":"CC0-1.0","has_fulltext":false,"has_references":false,"rights_class":"metadata_only","access":"paid","version":1,"content_endpoint":"/api/document/arxiv-2609.38867","fulltext_endpoint":null,"references_endpoint":null},{"id":"arxiv-2609.39028","title":"Improving Predicted MOS Scores, Not Perceived Quality: Multi-Predictor Test-Time Optimization of Enhanced Speech","year":2026,"authors":["Tsubasa Ochiai","Marc Delcroix","Nahomi Kusunoki"],"author_count":8,"journal":null,"doi":null,"source":"arxiv","topics":["speech-audio"],"keywords":["eess.AS","cs.SD"],"license":"CC0-1.0","has_fulltext":false,"has_references":false,"rights_class":"metadata_only","access":"paid","version":1,"content_endpoint":"/api/document/arxiv-2609.39028","fulltext_endpoint":null,"references_endpoint":null},{"id":"arxiv-2609.39030","title":"SURE-EVAL: A Systematic and Unified Agentic Framework for Reproducible Evaluation","year":2026,"authors":["Jing Peng","Junhao Du","Yixuan Wang"],"author_count":24,"journal":null,"doi":null,"source":"arxiv","topics":["speech-audio"],"keywords":["eess.AS"],"license":"CC0-1.0","has_fulltext":false,"has_references":false,"rights_class":"metadata_only","access":"paid","version":1,"content_endpoint":"/api/document/arxiv-2609.39030","fulltext_endpoint":null,"references_endpoint":null}]}