{
    "ok": true,
    "doi": "10.46243/jst.2019.v4.i06.pp42-50",
    "doi_display": "10.46243/jst.2019.v4.i06.pp42-50",
    "doi_url": "https://doi.org/10.46243/jst.2019.v4.i06.pp42-50",
    "state": "registered",
    "url": "https://www.jst.org.in/index.php/pub/article/view/98",
    "title": "BREAKING NEWS ARTICLE ANNOTATION USING IMAGE AND TEXT PROCESSING",
    "version": 2,
    "registered_via": "crossref",
    "prefix": {
        "prefix": "10.46243",
        "status": "live"
    },
    "registrant": {
        "name": "Longman Publishers",
        "kind": "publisher",
        "country": "India"
    },
    "reserved_at": null,
    "registered_at": "2026-09-29 22:00:41",
    "updated_at": "2026-09-30 00:00:01",
    "withdrawn_at": null,
    "withdrawn_reason": null,
    "record": {
        "format": "smartscholars-doi-metadata/1.0",
        "doi": "10.46243/jst.2019.v4.i06.pp42-50",
        "referent": "Creation",
        "type": "JournalArticle",
        "structural_type": "Digital",
        "modes": [
            "Visual"
        ],
        "characters": [
            "Language"
        ],
        "titles": [
            {
                "value": "BREAKING NEWS ARTICLE ANNOTATION USING IMAGE AND TEXT PROCESSING",
                "type": "PrincipalTitle",
                "lang": "en"
            }
        ],
        "identifiers": [
            {
                "type": "DOI",
                "value": "10.46243/jst.2019.v4.i06.pp42-50"
            }
        ],
        "agents": [
            {
                "role": "author",
                "name": {
                    "given": "",
                    "family": "U.GAYATHRI"
                },
                "sequence": "first"
            },
            {
                "role": "author",
                "name": {
                    "given": "",
                    "family": "M.V.BALARAM"
                },
                "sequence": "additional"
            },
            {
                "role": "publisher",
                "name": {
                    "org": "Longman Publishers"
                }
            }
        ],
        "dates": {
            "published": "2019",
            "date_type": "PublicationDate",
            "online": "2019"
        },
        "language": "en",
        "container": {
            "type": "Journal",
            "titles": [
                {
                    "value": "Journal of Science & Technology",
                    "type": "PrincipalTitle"
                }
            ],
            "identifiers": [
                {
                    "type": "ISSN",
                    "value": "2456-5660",
                    "medium": "electronic"
                }
            ],
            "volume": "04",
            "issue": "06",
            "pages": {
                "first": "42",
                "last": "50"
            }
        },
        "links": [
            {
                "url": "https://www.jst.org.in/index.php/pub/article/view/98",
                "return_type": "text/html",
                "primary": true
            },
            {
                "url": "https://www.jst.org.in/index.php/pub/article/download/98/84",
                "purpose": "text-mining",
                "return_type": "application/pdf"
            },
            {
                "url": "https://www.jst.org.in/index.php/pub/article/download/98/3308",
                "purpose": "text-mining",
                "return_type": "application/xml"
            },
            {
                "url": "https://www.jst.org.in/index.php/pub/article/view/98/84",
                "purpose": "similarity-checking"
            }
        ],
        "abstract": {
            "value": "Building upon recent Deep Neural Network architectures, current approaches lying in the intersection of Computer Vision and Natural Language Processing have achieved unprecedented breakthroughs in tasks like automatic captioning or image retrieval. Most of these learning methods, though, rely on large training sets of images associated with human annotations that specifically describe the visual content. In this paper we propose to go a step further and explore the more complex cases where textual descriptions are loosely related to the images. We focus on the particular domain of news articles in which the textual content often expresses connotative and ambiguous relations that are only suggested but not directly inferred from images. We introduce an adaptive CNN architecture that shares most of the structure for multiple tasks including source detection, article illustration and geolocation of articles. Deep Canonical Correlation Analysis is deployed for article illustration, and a new loss function based on Great Circle Distance is proposed for geolocation. Furthermore, we present BreakingNews, a novel dataset with approximately 100K news articles including images, text and captions, and enriched with heterogeneous meta-data (such as GPS coordinates and user comments). We show this dataset to be appropriate to explore all aforementioned problems, for which we provide a baseline performance using various Deep Learning architectures, and different representations of the textual and visual features. We report very promising results and bring to light several limitations of current state-of-the-art in this kind of domain, which we hope will help spur progress in the field.",
            "lang": "en"
        },
        "license": {
            "url": "https://creativecommons.org/licenses/by/4.0/",
            "start": "2019-01-01",
            "applies_to": "vor"
        },
        "references": [
            {
                "key": "ref1",
                "unstructured": "G. Andrew, R. Arora, J. Bilmes, and K. Livescu. Deep canonical correlation analysis. In ICML, 2013"
            },
            {
                "key": "ref2",
                "doi": "10.1109/iccv.2015.279",
                "unstructured": "Stanislaw Antol, Aishwarya Agrawal, Jiasen Lu, Margaret Mitchell, Dhruv Batra, C. Lawrence Zitnick, and Devi Parikh. VQA: Visual Question Answering. In International Conference on Computer Vision (ICCV), 2015"
            },
            {
                "key": "ref3",
                "unstructured": "K. Barnard, P. Duygulu, D. Forsyth, N. De Freitas, D. Blei, and M. Jordan. Matching words and pictures. The Journal of Machine Learning Research, 3:1107–1135, 2003"
            },
            {
                "key": "ref4",
                "doi": "10.1109/iccv.2001.937654",
                "unstructured": "K. Barnard and D. Forsyth. Learning the semantics of words and pictures. In ICCV, volume 2, pages 408–415. IEEE, 2001"
            },
            {
                "key": "ref5",
                "doi": "10.1145/1631272.1631292",
                "unstructured": "L. Cao, J. Yu, J. Luo, and T. Huang. Enhancing semantic and geographic annotation of web images via logistic canonical correlation regression. In ACM International Conference on Multimedia, pages 125–134. ACM, 2009"
            },
            {
                "key": "ref6",
                "doi": "10.3115/v1/w14-3102",
                "unstructured": "A. Chang, M. Savva, and C. Manning. Interactive learning of spatial knowledge for text to 3d scene generation. Sponsor: Idibon, page 14, 2014. Journal of Science and Technology ISSN: 2456-5660 Volume 4, Issue 06 (Nov-DEC 2019) www.jst.org.in DOI:https://doi.org/10.46243/jst.2019.v4.i06.pp42-50"
            },
            {
                "key": "ref7",
                "doi": "10.1109/cvpr.2011.5995610",
                "unstructured": "D. Chen, G. Baatz, K. Köser, S. Tsai, R. Vedantham, T. Pylvä, K. Roimela, X. Chen, J. Bach, M. Pollefeys, et al. City-scale landmark identification on mobile devices. In CVPR, pages 737–744. IEEE, 2011"
            },
            {
                "key": "ref8",
                "doi": "10.1109/cvpr.2015.7298856",
                "unstructured": "X. Chen and C. Zitnick. Mind’s eye: A recurrent visual representation for image caption generation. In CVPR, 2015"
            },
            {
                "key": "ref9",
                "doi": "10.1007/978-3-642-28997-2_28",
                "unstructured": "F. Coelho and C. Ribeiro. Image abstraction in crossmedia retrieval for text illustration. Lecture Notes in Computer Science, 7224 LNCS:329–339, 2012"
            },
            {
                "key": "ref10",
                "unstructured": "R. Collobert, J. Weston, L. Bottou, M. Karlen, K. Kavukcuoglu, and P. Kuksa. Natural language processing (almost) from scratch. JMLR, 12(08):2493–2537, 2011"
            },
            {
                "key": "ref11",
                "doi": "10.1145/383259.383316",
                "unstructured": "B. Coyne and R. Sproat. Wordseye: an automatic text-to-scene conversion system. In Conference on Computer Graphics and Interactive Techniques, pages 487–496. ACM, 2001"
            },
            {
                "key": "ref12",
                "doi": "10.1145/1526709.1526812",
                "unstructured": "D. Crandall, L. Backstrom, D. Huttenlocher, and J. Kleinberg. Mapping the world’s photos. In International Conference on World Wide Web, pages 761–770. ACM, 2009"
            }
        ],
        "record": {
            "registrant": "Longman Publishers",
            "registered": "2026-09-23",
            "updated": "2026-09-27",
            "issue_number": 1,
            "source": "crossref-api",
            "source_agency": "Crossref (member 25296)"
        }
    },
    "record_sha256": "31337f41d28441b3372f6207863888cdec72e6cc433c449b3cc589c436d90b6e",
    "handle": {
        "synced_at": "2026-10-01 18:14:03",
        "url": "https://www.jst.org.in/index.php/pub/article/view/98"
    },
    "links": {
        "record_page": "https://registry.smartscholars.in/record.php?doi=10.46243%2Fjst.2019.v4.i06.pp42-50",
        "system_metadata": "https://registry.smartscholars.in/resolve.php?doi=10.46243%2Fjst.2019.v4.i06.pp42-50&as=system",
        "history": "https://registry.smartscholars.in/api.php?action=history&doi=10.46243%2Fjst.2019.v4.i06.pp42-50",
        "kernel_xml": "https://registry.smartscholars.in/resolve.php?doi=10.46243%2Fjst.2019.v4.i06.pp42-50&as=xml"
    }
}