An Empirical Study on Utilizing Large Language Models for Bengali Image Caption Generation. Mahtab, M. A., Maisha, J., Rahman, M. M., & Kumar Saha Joy, S. In 2024 27th International Conference on Computer and Information Technology (ICCIT), pages 1714-1719, Dec, 2024. doi abstract bibtex An exemplary caption not only describes what is happening in a particular image but also denotes intricate traditional objects in the image by their local representative terms through which the native speakers can recognize the object in question. A caption that fails to accomplish the latter is not effective in conveying proper utility. To ensure caption locality, we aim to explore the potential of Large Language Models (LLMs) in Bengali image captioning, which have lately shown promising results in English language caption generation. As a first for the Bengali language, we utilized CLIP (Contrastive Language-Image Pre-training) encodings as a prefix to the captions by employing a mapping network, followed by fine-tuning BanglaGPT, a Bengali pre-trained large language model to generate the image captions. Furthermore, we explored vision transformer-based encoders (ViT, Swin) with BanglaGPT as the decoder. The best BanglaGPT-based model outperformed the current benchmark results, with BLEU-4, METEOR, and CIDEr scores of 54.3, 39.2, and 95.9 on the BanglaLekha dataset and 67.4, 36.6, and 76.9 on the BNature dataset.
@INPROCEEDINGS{11022443,
author={Mahtab, Muhammad Azmain and Maisha, Jannatim and Rahman, Md. Masudur and Kumar Saha Joy, Sajib},
booktitle={2024 27th International Conference on Computer and Information Technology (ICCIT)},
title={An Empirical Study on Utilizing Large Language Models for Bengali Image Caption Generation},
year={2024},
volume={},
number={},
pages={1714-1719},
abstract={An exemplary caption not only describes what is happening in a particular image but also denotes intricate traditional objects in the image by their local representative terms through which the native speakers can recognize the object in question. A caption that fails to accomplish the latter is not effective in conveying proper utility. To ensure caption locality, we aim to explore the potential of Large Language Models (LLMs) in Bengali image captioning, which have lately shown promising results in English language caption generation. As a first for the Bengali language, we utilized CLIP (Contrastive Language-Image Pre-training) encodings as a prefix to the captions by employing a mapping network, followed by fine-tuning BanglaGPT, a Bengali pre-trained large language model to generate the image captions. Furthermore, we explored vision transformer-based encoders (ViT, Swin) with BanglaGPT as the decoder. The best BanglaGPT-based model outperformed the current benchmark results, with BLEU-4, METEOR, and CIDEr scores of 54.3, 39.2, and 95.9 on the BanglaLekha dataset and 67.4, 36.6, and 76.9 on the BNature dataset.},
keywords={Computer vision;Image recognition;Image coding;Large language models;Computational modeling;Benchmark testing;Transformers;Decoding;Meteors;Information technology;Bengali image captioning;Large Language Model;CLIP;Vision Transformer;GPT-2;BanglaGPT},
doi={10.1109/ICCIT64611.2024.11022443},
ISSN={2474-9656},
month={Dec},}
Downloads: 0
{"_id":"NnfAjbtqApePTeBRr","bibbaseid":"mahtab-maisha-rahman-kumarsahajoy-anempiricalstudyonutilizinglargelanguagemodelsforbengaliimagecaptiongeneration-2024","author_short":["Mahtab, M. A.","Maisha, J.","Rahman, M. M.","Kumar Saha Joy, S."],"bibdata":{"bibtype":"inproceedings","type":"inproceedings","author":[{"propositions":[],"lastnames":["Mahtab"],"firstnames":["Muhammad","Azmain"],"suffixes":[]},{"propositions":[],"lastnames":["Maisha"],"firstnames":["Jannatim"],"suffixes":[]},{"propositions":[],"lastnames":["Rahman"],"firstnames":["Md.","Masudur"],"suffixes":[]},{"propositions":[],"lastnames":["Kumar","Saha","Joy"],"firstnames":["Sajib"],"suffixes":[]}],"booktitle":"2024 27th International Conference on Computer and Information Technology (ICCIT)","title":"An Empirical Study on Utilizing Large Language Models for Bengali Image Caption Generation","year":"2024","volume":"","number":"","pages":"1714-1719","abstract":"An exemplary caption not only describes what is happening in a particular image but also denotes intricate traditional objects in the image by their local representative terms through which the native speakers can recognize the object in question. A caption that fails to accomplish the latter is not effective in conveying proper utility. To ensure caption locality, we aim to explore the potential of Large Language Models (LLMs) in Bengali image captioning, which have lately shown promising results in English language caption generation. As a first for the Bengali language, we utilized CLIP (Contrastive Language-Image Pre-training) encodings as a prefix to the captions by employing a mapping network, followed by fine-tuning BanglaGPT, a Bengali pre-trained large language model to generate the image captions. Furthermore, we explored vision transformer-based encoders (ViT, Swin) with BanglaGPT as the decoder. The best BanglaGPT-based model outperformed the current benchmark results, with BLEU-4, METEOR, and CIDEr scores of 54.3, 39.2, and 95.9 on the BanglaLekha dataset and 67.4, 36.6, and 76.9 on the BNature dataset.","keywords":"Computer vision;Image recognition;Image coding;Large language models;Computational modeling;Benchmark testing;Transformers;Decoding;Meteors;Information technology;Bengali image captioning;Large Language Model;CLIP;Vision Transformer;GPT-2;BanglaGPT","doi":"10.1109/ICCIT64611.2024.11022443","issn":"2474-9656","month":"Dec","bibtex":"@INPROCEEDINGS{11022443,\n author={Mahtab, Muhammad Azmain and Maisha, Jannatim and Rahman, Md. Masudur and Kumar Saha Joy, Sajib},\n booktitle={2024 27th International Conference on Computer and Information Technology (ICCIT)}, \n title={An Empirical Study on Utilizing Large Language Models for Bengali Image Caption Generation}, \n year={2024},\n volume={},\n number={},\n pages={1714-1719},\n abstract={An exemplary caption not only describes what is happening in a particular image but also denotes intricate traditional objects in the image by their local representative terms through which the native speakers can recognize the object in question. A caption that fails to accomplish the latter is not effective in conveying proper utility. To ensure caption locality, we aim to explore the potential of Large Language Models (LLMs) in Bengali image captioning, which have lately shown promising results in English language caption generation. As a first for the Bengali language, we utilized CLIP (Contrastive Language-Image Pre-training) encodings as a prefix to the captions by employing a mapping network, followed by fine-tuning BanglaGPT, a Bengali pre-trained large language model to generate the image captions. Furthermore, we explored vision transformer-based encoders (ViT, Swin) with BanglaGPT as the decoder. The best BanglaGPT-based model outperformed the current benchmark results, with BLEU-4, METEOR, and CIDEr scores of 54.3, 39.2, and 95.9 on the BanglaLekha dataset and 67.4, 36.6, and 76.9 on the BNature dataset.},\n keywords={Computer vision;Image recognition;Image coding;Large language models;Computational modeling;Benchmark testing;Transformers;Decoding;Meteors;Information technology;Bengali image captioning;Large Language Model;CLIP;Vision Transformer;GPT-2;BanglaGPT},\n doi={10.1109/ICCIT64611.2024.11022443},\n ISSN={2474-9656},\n month={Dec},}\n\n\n\n","author_short":["Mahtab, M. A.","Maisha, J.","Rahman, M. M.","Kumar Saha Joy, S."],"key":"11022443","id":"11022443","bibbaseid":"mahtab-maisha-rahman-kumarsahajoy-anempiricalstudyonutilizinglargelanguagemodelsforbengaliimagecaptiongeneration-2024","role":"author","urls":{},"keyword":["Computer vision;Image recognition;Image coding;Large language models;Computational modeling;Benchmark testing;Transformers;Decoding;Meteors;Information technology;Bengali image captioning;Large Language Model;CLIP;Vision Transformer;GPT-2;BanglaGPT"],"metadata":{"authorlinks":{}}},"bibtype":"inproceedings","biburl":"https://bibbase.org/network/files/oiAweF3jCxzvQQS8F","dataSources":["jN2NaG6x9Dgv2w43Y"],"keywords":["computer vision;image recognition;image coding;large language models;computational modeling;benchmark testing;transformers;decoding;meteors;information technology;bengali image captioning;large language model;clip;vision transformer;gpt-2;banglagpt"],"search_terms":["empirical","study","utilizing","large","language","models","bengali","image","caption","generation","mahtab","maisha","rahman","kumar saha joy"],"title":"An Empirical Study on Utilizing Large Language Models for Bengali Image Caption Generation","year":2024}