{
    "ok": true,
    "doi": "10.46243/jst.2025.v10.i01.pp10-25",
    "doi_display": "10.46243/jst.2025.v10.i01.pp10-25",
    "doi_url": "https://doi.org/10.46243/jst.2025.v10.i01.pp10-25",
    "state": "registered",
    "url": "https://www.jst.org.in/index.php/pub/article/view/1136",
    "title": "Revolutionizing Audio Content Navigation: AI-Enhanced Multimodality and Machine Learning for Speaker Diarization and Topic Segmentation",
    "version": 2,
    "registered_via": "crossref",
    "prefix": {
        "prefix": "10.46243",
        "status": "live"
    },
    "registrant": {
        "name": "Longman Publishers",
        "kind": "publisher",
        "country": "India"
    },
    "reserved_at": null,
    "registered_at": "2026-09-29 22:00:32",
    "updated_at": "2026-09-29 23:59:51",
    "withdrawn_at": null,
    "withdrawn_reason": null,
    "record": {
        "format": "smartscholars-doi-metadata/1.0",
        "doi": "10.46243/jst.2025.v10.i01.pp10-25",
        "referent": "Creation",
        "type": "JournalArticle",
        "structural_type": "Digital",
        "modes": [
            "Visual"
        ],
        "characters": [
            "Language"
        ],
        "titles": [
            {
                "value": "Revolutionizing Audio Content Navigation: AI-Enhanced Multimodality and Machine Learning for Speaker Diarization and Topic Segmentation",
                "type": "PrincipalTitle",
                "lang": "en"
            }
        ],
        "identifiers": [
            {
                "type": "DOI",
                "value": "10.46243/jst.2025.v10.i01.pp10-25"
            }
        ],
        "agents": [
            {
                "role": "author",
                "name": {
                    "given": "",
                    "family": "Waseem Syed"
                },
                "sequence": "first"
            },
            {
                "role": "publisher",
                "name": {
                    "org": "Longman Publishers"
                }
            }
        ],
        "dates": {
            "published": "2025-01-25",
            "date_type": "PublicationDate",
            "online": "2025-01-25"
        },
        "language": "en",
        "container": {
            "type": "Journal",
            "titles": [
                {
                    "value": "Journal of Science & Technology",
                    "type": "PrincipalTitle"
                }
            ],
            "identifiers": [
                {
                    "type": "ISSN",
                    "value": "2456-5660",
                    "medium": "electronic"
                }
            ],
            "volume": "10",
            "issue": "1",
            "pages": {
                "first": "10",
                "last": "25"
            }
        },
        "links": [
            {
                "url": "https://www.jst.org.in/index.php/pub/article/view/1136",
                "return_type": "text/html",
                "primary": true
            },
            {
                "url": "https://jst.org.in/index.php/pub/article/download/1136/937",
                "purpose": "text-mining",
                "return_type": "application/pdf"
            }
        ],
        "abstract": {
            "value": "The rapid evolution of digital media has propelled an increasedconsumption of audio content, ranging from podcasts to educational lectures.Despite its growing popularity, the inherent unstructured nature of audio mediaposes significant challenges in navigation and user interaction. Our paperintroduces an innovative AI-driven framework designed to fundamentallytransform audio content exploration. Utilizing cutting-edge machine learning anddeep learning technologies, the system applies precise speaker diarization andtopic segmentation to radically improve navigation and content discovery inaudio streams. Furthermore, the incorporation of an interactive chat featureenriches user interaction, allowing listeners to effortlessly query and jumpdirectly to specific content via intuitive voice and text commands. This advancedsystem not only streamlines the audio exploration process but also personalizesthe listener experience by integrating multimodal interfaces and sophisticatedcontent annotation techniques. By addressing critical navigational inefficiencies,this framework sets a new paradigm in personalized, structured, and interactivemedia consumption, catering to the evolving demands of modern audio contentusers.",
            "lang": "en"
        },
        "license": {
            "url": "https://creativecommons.org/licenses/by/4.0",
            "start": "2025-01-25",
            "applies_to": "vor"
        },
        "references": [
            {
                "key": "ref1",
                "doi": "10.1109/tasl.2011.2125954",
                "unstructured": "Anguera X, Bozonnet S, Evans N, Fredouille C, Friedland G, Vinyals O. Speaker diarization: A review of recent research. IEEE Transactions on Audio, Speech, and Language Processing. 2012;20(2):356-370. doi:10.1109/ TASL.2011.2125954"
            },
            {
                "key": "ref2",
                "doi": "10.21437/interspeech.2020-3039",
                "unstructured": "Mao HH, Li S, McAuley J, Cottrell G. Speech recognition and multi-speaker diarization of long conversations. arXiv preprint. 2020;arXiv:2005.08072"
            },
            {
                "key": "ref3",
                "doi": "10.1080/10810730.2024.2321385",
                "unstructured": "Frølund, J., Løkke, A., Jensen, H., & Farver-Vestergaard, I. (2024). Development of Podcasts in a Hospital Setting: A User-Centered Approach. Journal of Health Communication, 29(4), 244–"
            },
            {
                "key": "ref4",
                "doi": "10.1080/10810730.2024.2321385",
                "unstructured": "https://doi.org/10.1080/10810730.2024. 2321385"
            },
            {
                "key": "ref5",
                "doi": "10.1109/tmm.2012.2233724",
                "unstructured": "Vallet F, Essid S, Carrive J. A multimodal approach to speaker diarization on TV talk-shows. IEEE Transactions on Multimedia. 2013;15(3):509-"
            },
            {
                "key": "ref6",
                "doi": "10.1109/tmm.2012.2233724",
                "unstructured": "doi:10.1109/TMM.2012.2233724"
            },
            {
                "key": "ref7",
                "doi": "10.1145/3643834.3661591",
                "unstructured": "Wang S, Ning Z, Truong A, et al. PodReels: Human-AI co-creation of video podcast teasers. Proceedings of the 2024 Designing Interactive Systems Conference (DIS ’24). ACM. 2024;17 pages. doi:10.1145/3643834.3661591"
            },
            {
                "key": "ref8",
                "doi": "10.1145/3626767.3625304",
                "unstructured": "Austin A, Samuel A. Enhancing podcasting by leveraging AI technologies. Communications of the ACM. 2023;66(10):48-55. doi:10.1145/3626767.3625304"
            },
            {
                "key": "ref9",
                "unstructured": "Edison Research. Podcast Consumer Report. Edison Research. 2023"
            },
            {
                "key": "ref10",
                "doi": "10.1016/j.specom.2009.08.009",
                "unstructured": "Kinnunen T, Li H. An overview of textindependent speaker recognition: From features to supervectors. Speech Communication. 2010;52(1):12-40. doi:10.1016/j. specom.2009.08.009"
            },
            {
                "key": "ref11",
                "doi": "10.1177/1461444820963776",
                "unstructured": "Chan-Olmsted, S., & Wang, R. (2022). Understanding podcast users: Consumption motives and behaviors. New Media & Society, 24(3), 684-704. https://doi. org/10.1177/1461444820963776"
            },
            {
                "key": "ref12",
                "doi": "10.1145/3173574.3174214",
                "unstructured": "Martin Porcheron, Joel E. Fischer, Stuart Reeves, and Sarah Sharples. 2018. Voice Interfaces in Everyday Life. In Proceedings of the 2018 CHI Conference on Human Factors in Computing Systems (CHI ‘18). Association for Computing Machinery, New York, NY, USA, Paper 640, 1–12. https://doi.org/10.1145/3173574.3174214"
            },
            {
                "key": "ref13",
                "doi": "10.1145/3591106.3592270",
                "unstructured": "Iacopo Ghinassi, Lin Wang, Chris Newell, and Matthew Purver. 2023. Multimodal Topic Segmentation of Podcast Shows with Pre-trained Neural Encoders. In Proceedings of the 2023 ACM International Conference on Multimedia Retrieval (ICMR ‘23). Association for Computing Machinery, New York, NY, USA, 602–606. https://doi.org/10.1145/3591106.3592270"
            },
            {
                "key": "ref14",
                "doi": "10.1109/ijcnn55064.2022.9892054",
                "unstructured": "A. Hajavi and A. Etemad, “Fine-grained Early Frequency Attention for Deep Speaker Recognition,” 2022 International Joint Conference on Neural Networks (IJCNN), Padua, Italy, 2022, pp. 1-6, doi: 10.1109/ IJCNN55064.2022.9892054"
            },
            {
                "key": "ref15",
                "doi": "10.1109/icpwc.2005.1431364",
                "unstructured": "U. S. Jha, “Efficient multimodal signal processing engine ease communication, computing, and multimedia convergence,” 2005 IEEE International Conference on Personal Wireless Communications, 2005. ICPWC 2005., New Delhi, India, 2005, pp. 348-352, doi: 10.1109/ ICPWC.2005.1431364"
            },
            {
                "key": "ref16",
                "doi": "10.3390/metrics1010002",
                "unstructured": "Galli, Carlo, et al. “Topic Modeling for Faster Literature Screening Using Transformer-Based Embeddings.” Metrics. Vol. 1. No 1. MDPI, 2024"
            },
            {
                "key": "ref17",
                "doi": "10.1109/tsa.2002.804546",
                "unstructured": "Lie Lu, Hong-Jiang Zhang and Hao Jiang, “Content analysis for audio classification and segmentation,” in IEEE Transactions on Speech and Audio Processing, vol. 10, no. 7, pp. 504- 516, Oct. 2002, doi: 10.1109/TSA.2002.804546"
            },
            {
                "key": "ref18",
                "doi": "10.1109/access.2022.3177584",
                "unstructured": "A. Gomez, M. S. Pattichis and S. Celedón-Pattichis, “Speaker Diarization and Identification From Single Channel Classroom Audio Recordings Using Virtual Microphones,” in IEEE Access, vol. 10, pp. 56256-56266, 2022, doi: 10.1109/ACCESS.2022.3177584"
            },
            {
                "key": "ref19",
                "doi": "10.1007/978-3-031-47718-8_7",
                "unstructured": "Ghosh, R. et al. (2024). Topic Segmentation of Semi-structured and Unstructured Conversational Datasets Using Language Models. In: Arai, K. (eds) Intelligent Systems and Applications. IntelliSys 2023. Lecture Notes in Networks and Systems, vol 825. Springer, Cham. https://doi.org/10.1007/978-3-031- 47718-8_7"
            },
            {
                "key": "ref20",
                "doi": "10.1145/3209811.3209875",
                "unstructured": "Deepika Yadav, Mayank Gupta, Malolan Chetlur, and Pushpendra Singh. 2018. Automatic Annotation of Voice Forum Content for Rural Users and Evaluation of Relevance. In Proceedings of the 1st ACM SIGCAS Conference on Computing and Sustainable Societies (COMPASS ‘18). Association for Computing Machinery, New York, NY, USA, Article 12, 1–11. https://doi.org/10.1145/3209811.3209875"
            },
            {
                "key": "ref21",
                "doi": "10.1109/icassp39728.2021.9414315",
                "unstructured": "F. Landini et al., “Analysis of the but Diarization System for Voxconverse Challenge,” ICASSP Waseem Syed: Revolutionizing Audio Content Navigation: AI-Enhanced Multimodality and Machine Learning for Speaker Diarization and Topic Segmentation 2021-2021 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP), Toronto, ON, Canada, 2021, pp. 5819-5823, doi: 10.1109/ICASSP39728.2021.9414315"
            },
            {
                "key": "ref22",
                "doi": "10.1007/978-981-19-3383-7_1",
                "unstructured": "Caratozzolo, P., Alvarez-Delgado, A., Hosseini, S. (2022). Natural Language Processing for Video Essays and Podcasts in Engineering. In: Hosseini, S., Peluffo, D.H., Nganji, J., Arrona-Palacios, A. (eds) Technology-Enabled Innovations in Education. Transactions on Computer Systems and Networks. Springer, Singapore. https://doi. org/10.1007/978-981-19-3383-7_1"
            },
            {
                "key": "ref23",
                "doi": "10.1038/s41746-022-00689-4",
                "unstructured": "Soenksen LR, Ma Y, Zeng C, Boussioux L, Villalobos Carballo K, Na L, Wiberg HM, Li ML, Fuentes I, Bertsimas D. Integrated multimodal artificial intelligence framework for healthcare applications. NPJ Digit Med. 2022 Sep 20;5(1):149. doi: 10.1038/s41746-022-00689-"
            },
            {
                "key": "ref24",
                "unstructured": "PMID: 36127417; PMCID: PMC9489871"
            },
            {
                "key": "ref25",
                "doi": "10.1109/mcas.2019.2945210",
                "unstructured": "Y. Zhao, X. Xia and R. Togneri, “Applications of Deep Learning to Audio Generation,” in IEEE Circuits and Systems Magazine, vol. 19, no. 4, pp. 19-38, Fourthquarter 2019, doi: 10.1109/ MCAS.2019.2945210"
            },
            {
                "key": "ref26",
                "doi": "10.1109/tvcg.2024.3456189",
                "unstructured": "M. R. Mahmud, A. Cordova and J. Quarles, “Multimodal Feedback Methods for Advancing the Accessibility of Immersive Virtual Reality for People With Balance Impairments Due to Multiple Sclerosis,” in IEEE Transactions on Visualization and Computer Graphics, vol. 30, no. 11, pp. 7193-7202, Nov. 2024, doi: 10.1109/ TVCG.2024.3456189"
            },
            {
                "key": "ref27",
                "unstructured": "Edison Research: Sound-Data-The-State-of-Audio-in-50-Charts. https:// www.edisonresearch.com/wp-content/ uploads/2024/06/Sound-Data-The-State-of-Audio-in-50-Charts.pdf"
            }
        ],
        "record": {
            "registrant": "Longman Publishers",
            "registered": "2025-08-26",
            "updated": "2026-09-27",
            "issue_number": 1,
            "source": "crossref-api",
            "source_agency": "Crossref (member 25296)"
        }
    },
    "record_sha256": "ab91e8b0c772986b2563631e4262136fa2973ac95d4240fd7afcbb417d678c86",
    "handle": {
        "synced_at": "2026-10-01 18:12:53",
        "url": "https://www.jst.org.in/index.php/pub/article/view/1136"
    },
    "links": {
        "record_page": "https://registry.smartscholars.in/record.php?doi=10.46243%2Fjst.2025.v10.i01.pp10-25",
        "system_metadata": "https://registry.smartscholars.in/resolve.php?doi=10.46243%2Fjst.2025.v10.i01.pp10-25&as=system",
        "history": "https://registry.smartscholars.in/api.php?action=history&doi=10.46243%2Fjst.2025.v10.i01.pp10-25",
        "kernel_xml": "https://registry.smartscholars.in/resolve.php?doi=10.46243%2Fjst.2025.v10.i01.pp10-25&as=xml"
    }
}