Home / Current Issue / Paper 1711964
Cross-Modal Synthesis: Generating Semantic Segmentation Masks from Image Captions and Vice-Versa using Multi-Modal Transformers
Subject area: Science,Engineering and Technology · Area of research: Computer Vision and Natural Language Processing
DOI: https://doi.org/10.64388/IREV8I11-1711964
Abstract
The integration of visual and linguistic modalities has transformed computer vision and natural language processing research. Cross-modal synthesis, which seeks to generate segmentation masks from text and captions from images, presents significant opportunities for understanding visual scenes semantically. In this work, we propose a unified transformer-based framework that performs bi-directional synthesis between image captions and semantic segmentation masks. The model, built on top of a multi-modal transformer encoder-decoder, learns shared latent representations enabling seamless translation between visual regions and linguistic tokens. Extensive experiments on COCO-Stuff and ADE20K datasets demonstrate that our method outperforms baseline models by 16% in mIoU for caption-to-mask synthesis and 14% BLEU improvement for mask-to-caption generation, establishing a new benchmark for multi-modal reasoning.
How to cite this paper
@article{1711964,
author = {Subham Sahoo, Rajat Gupta, Chintan Tibrewala},
title = {Cross-Modal Synthesis: Generating Semantic Segmentation Masks from Image Captions and Vice-Versa using Multi-Modal Transformers},
journal = {Iconic Research And Engineering Journals},
year = {2025},
volume = {8},
number = {11},
pages = {2418-2423},
issn = {2456-8880},
url = {https://www.irejournals.com/formatedpaper/1711964.pdf},
abstract = {The integration of visual and linguistic modalities has transformed computer vision and natural language processing research. Cross-modal synthesis, which seeks to generate segmentation masks from text and captions from images, presents significant opportunities for understanding visual scenes semantically. In this work, we propose a unified transformer-based framework that performs bi-directional synthesis between image captions and semantic segmentation masks. The model, built on top of a multi-modal transformer encoder-decoder, learns shared latent representations enabling seamless translation between visual regions and linguistic tokens. Extensive experiments on COCO-Stuff and ADE20K datasets demonstrate that our method outperforms baseline models by 16% in mIoU for caption-to-mask synthesis and 14% BLEU improvement for mask-to-caption generation, establishing a new benchmark for multi-modal reasoning.},
month = {May},
doi = {https://doi.org/10.64388/IREV8I11-1711964}
}