<?xml version="1.0" encoding="UTF-8"?>
<!DOCTYPE article PUBLIC "-//NLM//DTD JATS (Z39.96) Journal Publishing DTD v1.1 20151215//EN" "http://jats.nlm.nih.gov/publishing/1.1/JATS-journalpublishing1.dtd">
<article xmlns:xlink="http://www.w3.org/1999/xlink" xmlns:mml="http://www.w3.org/1998/Math/MathML" xml:lang="en" article-type="research-article" dtd-version="1.1">
<front>
<journal-meta>
<journal-id journal-id-type="pmc">CMES</journal-id>
<journal-id journal-id-type="nlm-ta">CMES</journal-id>
<journal-id journal-id-type="publisher-id">CMES</journal-id>
<journal-title-group>
<journal-title>Computer Modeling in Engineering &#x0026; Sciences</journal-title>
</journal-title-group>
<issn pub-type="epub">1526-1506</issn>
<issn pub-type="ppub">1526-1492</issn>
<publisher>
<publisher-name>Tech Science Press</publisher-name>
<publisher-loc>USA</publisher-loc>
</publisher>
</journal-meta>
<article-meta>
<article-id pub-id-type="publisher-id">58456</article-id>
<article-id pub-id-type="doi">10.32604/cmes.2025.058456</article-id>
<article-categories>
<subj-group subj-group-type="heading">
<subject>Article</subject>
</subj-group>
</article-categories>
<title-group>
<article-title>Integrating Speech-to-Text for Image Generation Using Generative Adversarial Networks</article-title>
<alt-title alt-title-type="left-running-head">Integrating Speech-to-Text for Image Generation Using Generative Adversarial Networks</alt-title>
<alt-title alt-title-type="right-running-head">Integrating Speech-to-Text for Image Generation Using Generative Adversarial Networks</alt-title>
</title-group>
<contrib-group>
<contrib id="author-1" contrib-type="author">
<name name-style="western"><surname>Mahajan</surname><given-names>Smita</given-names></name><xref ref-type="aff" rid="aff-1">1</xref></contrib>
<contrib id="author-2" contrib-type="author">
<name name-style="western"><surname>Gite</surname><given-names>Shilpa</given-names></name><xref ref-type="aff" rid="aff-1">1</xref><xref ref-type="aff" rid="aff-2">2</xref></contrib>
<contrib id="author-3" contrib-type="author" corresp="yes">
<name name-style="western"><surname>Pradhan</surname><given-names>Biswajeet</given-names></name><xref ref-type="aff" rid="aff-3">3</xref><xref rid="cor1" ref-type="corresp">&#x002A;</xref><email>biswajeet.pradhan@uts.edu.au</email></contrib>
<contrib id="author-4" contrib-type="author">
<name name-style="western"><surname>Alamri</surname><given-names>Abdullah</given-names></name><xref ref-type="aff" rid="aff-4">4</xref></contrib>
<contrib id="author-5" contrib-type="author">
<name name-style="western"><surname>Inamdar</surname><given-names>Shaunak</given-names></name><xref ref-type="aff" rid="aff-5">5</xref></contrib>
<contrib id="author-6" contrib-type="author">
<name name-style="western"><surname>Shriyansh</surname><given-names>Deva</given-names></name><xref ref-type="aff" rid="aff-5">5</xref></contrib>
<contrib id="author-7" contrib-type="author">
<name name-style="western"><surname>Shah</surname><given-names>Akshat Ashish</given-names></name><xref ref-type="aff" rid="aff-5">5</xref></contrib>
<contrib id="author-8" contrib-type="author">
<name name-style="western"><surname>Agarwal</surname><given-names>Shruti</given-names></name><xref ref-type="aff" rid="aff-5">5</xref></contrib>
<aff id="aff-1"><label>1</label><institution>Artificial Intelligence and Machine Learning Department, Symbiosis Institute of Technology</institution>, <addr-line>Pune, 412115</addr-line>, <country>India</country></aff>
<aff id="aff-2"><label>2</label><institution>Symbiosis Centre of Applied AI (SCAAI), Symbiosis Institute of Technology</institution>, <addr-line>Pune, 412115</addr-line>, <country>India</country></aff>
<aff id="aff-3"><label>3</label><institution>Centre for Advanced Modelling and Geospatial Information Systems (CAMGIS), School of Civil and Environmental Engineering, University of Technology Sydney</institution>, <addr-line>Sydney, NSW 2007</addr-line>, <country>Australia</country></aff>
<aff id="aff-4"><label>4</label><institution>Department of Geology and Geophysics, College of Science, King Saud University</institution>, <addr-line>Riyadh, 11451</addr-line>, <country>Saudi Arabia</country></aff>
<aff id="aff-5"><label>5</label><institution>Department of Computer Science and Engineering, Symbiosis Institute of Technology</institution>, <addr-line>Pune, 412115</addr-line>, <country>India</country></aff>
</contrib-group>
<author-notes>
<corresp id="cor1"><label>&#x002A;</label>Corresponding Author: Biswajeet Pradhan. Email: <email>biswajeet.pradhan@uts.edu.au</email></corresp>
</author-notes>
<pub-date date-type="collection" publication-format="electronic">
<year>2025</year>
</pub-date>
<pub-date date-type="pub" publication-format="electronic">
<day>30</day><month>05</month><year>2025</year>
</pub-date>
<volume>143</volume>
<issue>2</issue>
<fpage>2001</fpage>
<lpage>2026</lpage>
<history>
<date date-type="received">
<day>12</day>
<month>9</month>
<year>2024</year>
</date>
<date date-type="accepted">
<day>26</day>
<month>1</month>
<year>2025</year>
</date>
</history>
<permissions>
<copyright-statement>&#x00A9; 2025 The Authors.</copyright-statement>
<copyright-year>2025</copyright-year>
<copyright-holder>Published by Tech Science Press.</copyright-holder>
<license xlink:href="https://creativecommons.org/licenses/by/4.0/">
<license-p>This work is licensed under a <ext-link ext-link-type="uri" xlink:type="simple" xlink:href="https://creativecommons.org/licenses/by/4.0/">Creative Commons Attribution 4.0 International License</ext-link>, which permits unrestricted use, distribution, and reproduction in any medium, provided the original work is properly cited.</license-p>
</license>
</permissions>
<self-uri content-type="pdf" xlink:href="TSP_CMES_58456.pdf"></self-uri>
<abstract>
<p>The development of generative architectures has resulted in numerous novel deep-learning models that generate images using text inputs. However, humans naturally use speech for visualization prompts. Therefore, this paper proposes an architecture that integrates speech prompts as input to image-generation Generative Adversarial Networks (GANs) model, leveraging Speech-to-Text translation along with the CLIP &#x002B; VQGAN model. The proposed method involves translating speech prompts into text, which is then used by the Contrastive Language-Image Pretraining (CLIP) &#x002B; Vector Quantized Generative Adversarial Network (VQGAN) model to generate images. This paper outlines the steps required to implement such a model and describes in detail the methods used for evaluating the model. The GAN model successfully generates artwork from descriptions using speech and text prompts. Experimental outcomes of synthesized images demonstrate that the proposed methodology can produce beautiful abstract visuals containing elements from the input prompts. The model achieved a Fr&#x00E9;chet Inception Distance (FID) score of 28.75, showcasing its capability to produce high-quality and diverse images. The proposed model can find numerous applications in educational, artistic, and design spaces due to its ability to generate images using speech and the distinct abstract artistry of the output images. This capability is demonstrated by giving the model out-of-the-box prompts to generate never-before-seen images with plausible realistic qualities.</p>
</abstract>
<kwd-group kwd-group-type="author">
<kwd>Generative adversarial networks</kwd>
<kwd>speech-to-image translation</kwd>
<kwd>visualization</kwd>
<kwd>transformers</kwd>
<kwd>prompt engineering</kwd>
</kwd-group>
<funding-group>
<award-group id="awg1">
<funding-source>King Saud University</funding-source>
<award-id>ORF-2025-14</award-id>
</award-group>
</funding-group>
</article-meta>
</front>
<body>
<sec id="s1">
<label>1</label>
<title>Introduction</title>
<p>Image generation from speech inputs has been a fascinating research topic but comes with challenges of accuracy of the generated images after translation of audio into pictorial output. It is based on the concept of Generative Adversarial Networks (GANs) which have two different neural networks namely generator and discriminator. These two neural networks continuously improve their performances based on each other&#x2019;s feedback. The generator aims to create realistic content so that the discriminator would be fooled by the newly generated synthetic content [<xref ref-type="bibr" rid="ref-1">1</xref>]. GANs are primarily used in image generation applications but can be expanded in other domains like audio, video, code, and even creative content as well.</p>
<p>Image generation using speech as input is essential and interesting because it more natural interaction between the system and the human [<xref ref-type="bibr" rid="ref-2">2</xref>]. In traditional approaches, text is normally used as an input, but it is not easier, natural interaction when compared with speech. One more advantage of using speech over text is identifying emotions and nuances with speech which gives added expressions and a comprehensive image-generation experience [<xref ref-type="bibr" rid="ref-3">3</xref>].</p>
<p>Natural Language Processing (NLP) has been considered one of the ground-breaking topics in the last few years. With the advancements in transformer architectures, there are magical milestones like GPT3 with billions of parameters and considered as one of the most popular transformers [<xref ref-type="bibr" rid="ref-4">4</xref>]. But translating detailed narratives into realistic pictures is a difficult task. While designs like Stacked Generative Adversarial Networks (StackGAN&#x002B;&#x002B;) [<xref ref-type="bibr" rid="ref-5">5</xref>] have shown promising results in recent years [<xref ref-type="bibr" rid="ref-6">6</xref>], they have been limited by the visual aspects of their training datasets. More recently, OpenAI introduced a groundbreaking deep network named Contrastive Language-Image Pretraining (CLIP) [<xref ref-type="bibr" rid="ref-7">7</xref>]. The CLIP system comprises two encoders [<xref ref-type="bibr" rid="ref-8">8</xref>], one for text and one for images. Through pretraining on 400,000,000 image-text pairs (consisting of images and their corresponding captions), CLIP learns to generate comparable embeddings for words and pictures that convey related concepts [<xref ref-type="bibr" rid="ref-9">9</xref>]. The versatility of CLIP extends beyond training, making it applicable to visual categorization tasks without the need for further training. CLIP exhibits impressive &#x201C;zero-shot&#x201D; capabilities, enabling it to accurately predict entire classes it has never encountered before. When a model attempts to predict a class, it has only been seen once in the training data; this is referred to as &#x201C;zero-shot learning&#x201D;. Models like CLIP typically excel at zero-shot learning due to their utilization of text information in (image, text) pairs. Even when presented with significantly different images from those in the training set, the CLIP model can often provide a reliable caption for the image [<xref ref-type="bibr" rid="ref-10">10</xref>].</p>
<p>In this research, the authors propose a CLIP-based model designed to generate the best matching image for a text embedding obtained from a speech input as shown in <xref ref-type="fig" rid="fig-1">Fig. 1</xref>. This input can be used for various purposes, including producing samples for image datasets, snaps of human faces, representative photographs, cartoon or animated characters, image-to-image translation, text-to-image translation, semantic-image-to-photo translation [<xref ref-type="bibr" rid="ref-10">10</xref>], generating photo-realistic paintings using ExGANs [<xref ref-type="bibr" rid="ref-11">11</xref>], face frontal view generation [<xref ref-type="bibr" rid="ref-11">11</xref>], generating new human poses [<xref ref-type="bibr" rid="ref-12">12</xref>], photos to emojis, photograph editing [<xref ref-type="bibr" rid="ref-13">13</xref>], face aging, photo blending [<xref ref-type="bibr" rid="ref-14">14</xref>], super-resolution, photo inpainting, clothing translation, video prediction, and 3D object generation [<xref ref-type="bibr" rid="ref-15">15</xref>]. While current models have embraced a multi-modal approach by incorporating various inputs into their algorithms, the use of speech remains unknown in image generation. A speech-driven generative engine holds numerous promising applications in the realm of virtual reality. However, there has been limited progress in developing a speech-based GAN encoder. The most recent and significant advancement in this direction has been pioneered by Goodfe et al. [<xref ref-type="bibr" rid="ref-16">16</xref>].</p>
<fig id="fig-1">
<label>Figure 1</label>
<caption>
<title>Transforming speech to image</title>
</caption>
<graphic mimetype="image" mime-subtype="tif" xlink:href="CMES_58456-fig-1.tif"/>
</fig>
<p>In this paper, a speech-to-image generation model is presented using Vector-Quantized GANs and the CLIP Transformer [<xref ref-type="bibr" rid="ref-17">17</xref>]. The novelty of this approach is to integrate speech as a prompt for generating pictures. These speech descriptions are translated into text embeddings and inputted into the generative model to produce plausible imagery as outputs. The proposed model uses GAN along with the CLIP transformer and Vector Quantized Generative Adversarial Network (VQ-GAN) model, using speech-to-text conversion [<xref ref-type="bibr" rid="ref-18">18</xref>]. The authors aim to develop technology that allows us to visualize human speech, which could have numerous applications in education and artwork [<xref ref-type="bibr" rid="ref-19">19</xref>]. For example, GANs can be used to create various environments and levels in games as well as to add new levels to existing games. The Non-fungible Tokens&#x2019; (NFT) space, where abstract AI-generated art has seen significant popularity in recent years and gaining attention [<xref ref-type="bibr" rid="ref-20">20</xref>].</p>
<p>CLIP has the potential of cross-modal assimilation useful for text-to-image conversion [<xref ref-type="bibr" rid="ref-21">21</xref>]. It is easier to capture the corresponding descriptions while CLIP models are implemented. On the other hand, VQGAN [Vector Quantized Generative Adversarial Network] enhances resolution and control over detail, creating high-quality and systematic image synthesis by building images as discrete representations [<xref ref-type="bibr" rid="ref-21">21</xref>]. VQGAN allows for high-quality and structured image synthesis by rendering images as discrete representations, improving resolution and detail control. CLIP &#x002B; VQGAN can effectively align text with image features due to CLIP&#x2019;s pre-trained language-vision model, transcending conventional GANs (like StackGAN and AttnGAN) floundering with subtle language cues. Earlier models sometimes miss minor textual features, particularly for conceptual descriptions or complicated settings, as they depend on the latent space method of CLIP. With excellent control over subtle details, VQGAN&#x2019;s quantization technique produces high-quality, high-resolution images comparable to or better than StackGAN&#x002B;&#x002B;. The resolution capabilities of prior models, such as AttnGAN and StackGAN&#x002B;&#x002B;, are restricted. VQGAN is suitable for high-resolution necessities since it creates images with exceptional scaling and structure. CLIP &#x002B; VQGAN can produce and comprehend visuals more effectively, even with unclear text, due to CLIP&#x2019;s natural strong language understanding. Earlier methods used by DM-GAN did not achieve CLIP&#x2019;s cross-modal understanding depth but added dynamic memory to handle weak text definitions. The pre-trained language processing of CLIP suggests a reliable basis for flexible and coherent outputs. In contrast, previous models need extra modules to adapt to ambiguous or multi-meaning text inputs. CLIP&#x2019;s inherent text-image embedding alignment can reduce computing requirements, making it more effective and versatile than AttnGAN, which extensively uses attention processes. Prior models such as AttnGAN that heavily rely on attention may present computational difficulties, mainly when negotiating with intricate scenes. This prerequisite is reduced, and efficiency is increased by using CLIP. The CLIP &#x002B; VQGAN model significantly gains flexibility, text-image alignment, and detail quality [<xref ref-type="bibr" rid="ref-21">21</xref>]. Compared to earlier GAN-based models, this model is especially effective for applications that need computing efficiency, high-resolution imagery, and subtle language interpretation.</p>
<p>GANs and NFTs converge on the concept of creativity, with GANs demonstrating creativity through complex algorithms trained on datasets, like how humans draw inspiration from existing artworks [<xref ref-type="bibr" rid="ref-22">22</xref>]. In recent years, recurrent neural networks and GANs have made significant progress in zero-shot recognition and image translation research [<xref ref-type="bibr" rid="ref-23">23</xref>]. The BigGANs are used to generate synthetic photographs that are nearly indistinguishable from the originals. Additionally, StackGANs can be used to translate realistic images based on text-based descriptions.</p>
<p>The novelty of our approach lies in the integration of Speech-to-Text translation with CLIP &#x002B; VQGAN for image generation. Unlike existing models that primarily use text or image inputs, our model leverages the auditory modality, providing a unique and innovative method for generating images. This technique of generative models is impactful and also enhances their applicability in various fields. While previous works have explored text-to-image generation using GANs and transformer models, the incorporation of speech input as a direct prompt for image synthesis is a novel and unexplored approach. This speech-driven image generation approach has potential applications in various fields, including education, artwork creation, gaming, and virtual reality environments. The paper showcases significant results and low FID scores of 31 to 65 based on multiple examples of generated images across various topics and domains.</p>
<p>Overall, the main contribution of the paper can be summarized as:
<list list-type="bullet">
<list-item>
<p>Integration of CLIP for Text-Image Conversion: The architecture incorporates CLIP to strengthen the alignment between text and image features.</p></list-item>
<list-item>
<p>Novelty in Text-to-Image Generation: This approach introduces a new modality by using speech prompts as inputs for text-to-image generation.</p></list-item>
<list-item>
<p>Combination of Speech Input with CLIP, GAN, and Iterative Latent Space: The model integrates speech inputs with CLIP, GAN, and an iterative latent space process, enhancing the generation process.</p></list-item>
</list></p>
<p>This paper delves into an in-depth literature review presenting the history of popular GAN models, transformer-based models, and text-to-image translation models implemented using generative architectures in <xref ref-type="sec" rid="s2">Sections 2</xref>&#x2013;<xref ref-type="sec" rid="s4">4</xref>. Subsequently, <xref ref-type="sec" rid="s5">Section 5</xref> highlights the proposed architectural model of this paper, which utilizes speech input prompts for generating images, and the datasets used are discussed. <xref ref-type="sec" rid="s6">Section 6</xref> is dedicated to discussing the results, and <xref ref-type="sec" rid="s7">Section 7</xref> states the conclusion and addresses the findings of this concept.</p>
</sec>
<sec id="s2">
<label>2</label>
<title>Literature Review</title>
<p>GANs involve two deep learning networks locked in a competitive training process. The first network, the generator, acts like a creative artist, trying to produce realistic images. The second, the discriminator, plays the role of a critical art reviewer, aiming to distinguish real images from the generator&#x2019;s creations.</p>
<p>The focused literature review presented here provides context specific to the GAN variants or modifications used, particularly when the work builds upon traditional GAN concepts in innovative ways. The authors have included this review to focus on context, model evolution, and relevance to the study, even though foundational GAN information is widely accessible. This review enables the authors to highlight how these adaptations address specific challenges or limitations in existing GAN models, reinforcing the importance of the chosen methodology. It offers a concise background and prepares readers for the technical details of the methodology and experiments that follow.</p>
<sec id="s2_1">
<label>2.1</label>
<title>Deep Convolutional Generative Adversarial Network (DCGAN)</title>
<p>Applying the Transposed Convolution Operation on GANs acts as an upscaling operation that helps translate lower-resolution images to higher-resolution images [<xref ref-type="bibr" rid="ref-24">24</xref>]. Conditional Generative Adversarial Network (CGAN) CGAN solves the problem of data generated from noise found in GANs (especially in images with multiple classes) by making sure the generator is creating images only of one particular class [<xref ref-type="bibr" rid="ref-25">25</xref>]. CycleGAN: CycleGAN aims to solve the problem of image-to-image translation. CycleGAN performs the unpaired image-to-image translation. That means the images which are used for training, do not have to represent the same thing [<xref ref-type="bibr" rid="ref-26">26</xref>]. Coupled Generative Adversarial Networks (CoGAN) To confuse the discriminative models, a group of generative models collaborates to synthesize a pair of pictures from two distinct domains. The discriminative models seek to distinguish between pictures derived from the distribution of training data in the relevant domains and those produced from the corresponding generative models [<xref ref-type="bibr" rid="ref-27">27</xref>].</p>
<p>ProGAN Progressive growing of Generative Adversarial Networks One of the issues with GANs can be traced back to instability in Training. The loss of the GAN can occasionally oscillate as the learning of one by the generator and the other by the discriminator is undone. In other cases, the loss may materialize immediately after the networks converge and the photos begin to appear appalling. ProGAN, or the progressive expansion of generative adversarial networks, is a method for stabilizing GAN training by gradually boosting the produced image&#x2019;s resolution [<xref ref-type="bibr" rid="ref-28">28</xref>]. WGAN (Wasserstein Generative Adversarial Networks): This minimizes an approximation of the Earth-Mover&#x2019;s distance [EM] rather than the Jensen-Shannon divergence as in the original GAN formulation [<xref ref-type="bibr" rid="ref-29">29</xref>]. SAGAN Self-Attention Generative Adversarial Networks: For long-range dependency modeling for image generation tasks, the SelfAttention Generative Adversarial Network, or SAGAN, is used. Only spatially local points in lower-resolution feature maps are used by conventional convolutional GANs to produce high-resolution details. Details in SAGAN may be produced utilizing hints from all feature locations [<xref ref-type="bibr" rid="ref-30">30</xref>]. Additionally, the discriminator may verify the consistency of extremely detailed features in distant areas of the picture [<xref ref-type="bibr" rid="ref-31">31</xref>]. Big Generative Adversarial Networks (BigGAN) The BigGAN method combines a variety of current best practices for training class-conditional pictures while scaling up batch size and model parameter numbers. As a result, photographs with high resolution (big size) and excellent quality (high fidelity) are often produced [<xref ref-type="bibr" rid="ref-32">32</xref>]. Style-based Generative Adversarial Networks (StyleGAN), the Style Generative Adversarial Network, also known as StyleGAN, is an addition to the GAN architecture that suggests significant changes to the generator model. These changes include the use of a mapping network to map points in latent space to an intermediate latent space, the use of the intermediate latent space to control style at each point in the generator model, and the addition of noise as a source of variation at each point in the generator model [<xref ref-type="bibr" rid="ref-33">33</xref>].</p>
</sec>
<sec id="s2_2">
<label>2.2</label>
<title>Generative Adversarial Networks</title>
<p>Generative Adversarial Networks (GANs) are used in generating image, video, and voice data. They are trained to utilize two neural network models: a generator that learns to produce new data and a discriminator which distinguishes between real and fake data generated by the generator. The generator aims to deceive the discriminator, while the discriminator possesses all pertinent information. Following training, generative models can be utilized to generate large volumes of data as required [<xref ref-type="bibr" rid="ref-10">10</xref>]. Creating visuals from detailed captions has been a challenge that researchers and developers have previously addressed. However, the introduction of GANs has significantly raised the performance of this task. Designs like StackGAN&#x002B;&#x002B; [<xref ref-type="bibr" rid="ref-34">34</xref>] and align (Deep Recurrent Attentive Writer) DRAW [<xref ref-type="bibr" rid="ref-35">35</xref>] have demonstrated promising results in recent years despite being constrained by the visual and textual domains of the training dataset. Additionally, other applications, such as image synthesis and style transfer, have evolved from the two-player methodology introduced by Ian Goodfellow in 2014. More recent Text-to-Image translation applications are implemented using various GAN-based generative models, including Variational Auto Encoder and transformers [<xref ref-type="bibr" rid="ref-36">36</xref>,<xref ref-type="bibr" rid="ref-37">37</xref>]. Latest research papers also discussed stable diffusion models for text-to-image generation with transformer-based architectures as shown in table [<xref ref-type="bibr" rid="ref-38">38</xref>,<xref ref-type="bibr" rid="ref-39">39</xref>]. GAN variants (e.g., ProGAN, StyleGAN) are primarily developed for high-quality image synthesis (e.g., realistic faces, objects), their architectural innovations and techniques have significantly influenced text-to-image generation models. Text-to-image models like DALL&#x00B7;E and Stable Diffusion also require generating high-resolution images. The progressive growing strategy inspired techniques to scale resolution in text-to-image pipelines while maintaining detail. Style GAN&#x2019;s ability to separate content (structure) and style (appearance) has direct implications for text-to-image generation, where the model must interpret a text description and translate it into distinct but coherent visual features [<xref ref-type="bibr" rid="ref-40">40</xref>,<xref ref-type="bibr" rid="ref-41">41</xref>].</p>
</sec>
</sec>
<sec id="s3">
<label>3</label>
<title>Working of Stack GAN</title>
<p>Stacked GANS is widely used for producing high-resolution photos with numerous photorealistic features [<xref ref-type="bibr" rid="ref-42">42</xref>]. The text-to-image generation process is mainly divided into two parts [<xref ref-type="bibr" rid="ref-43">43</xref>] as shown in <xref ref-type="fig" rid="fig-2">Fig. 2</xref>.</p>
<fig id="fig-2">
<label>Figure 2</label>
<caption>
<title>Functioning of stack-generative adversarial networks</title>
</caption>
<graphic mimetype="image" mime-subtype="tif" xlink:href="CMES_58456-fig-2.tif"/>
</fig>
<sec id="s3_1">
<label>3.1</label>
<title>Stage 1 GAN</title>
<p>A basic low-resolution image is created by outlining the basic colors and shape of the object according to the provided text description. Subsequently, the model constructs the background layout using a random noise vector. The GAN then sketches the primitive shape and colors of a scene based on the provided text description, resulting in low-resolution images.</p>
</sec>
<sec id="s3_2">
<label>3.2</label>
<title>Stage 2 GAN</title>
<p>It completes the object&#x2019;s details by reevaluating the text description. It rectifies any errors present in the low-resolution image from Stage 1, thereby creating a high-resolution photo-realistic image [<xref ref-type="bibr" rid="ref-44">44</xref>]. This stage utilizes the results from Stage 1 along with the text description as inputs to generate high-resolution images with photo-realistic details. These stages are explored in depth by Li et al. [<xref ref-type="bibr" rid="ref-28">28</xref>].</p>
<p><bold>Functioning of Stack Generative Adversarial Networks StackGAN:</bold> This approach utilizes a two-stage GAN system called StackGAN to generate images based on text descriptions [<xref ref-type="bibr" rid="ref-45">45</xref>].</p>
<p><bold>Stage 1: Capturing the Essence</bold></p>
<p><list list-type="bullet">
<list-item>
<p><bold>Text Encoding</bold></p></list-item>
</list></p>
<p>Translating the textual description (e.g., a sentence describing an object) into a concise representation using techniques like word embeddings [<xref ref-type="bibr" rid="ref-5">5</xref>] or recurrent neural networks (RNNs) [<xref ref-type="bibr" rid="ref-46">46</xref>]. Think of this as capturing the core meaning of the words.
<list list-type="bullet">
<list-item>
<p><bold>Injecting Creativity with Noise</bold></p></list-item>
</list></p>
<p>This encoded text is then combined with random noise, which acts as a spark of creativity for the generator network. Imagine adding some artistic flair to the textual blueprint.
<list list-type="bullet">
<list-item>
<p><bold>Low-Resolution Sketch</bold></p></list-item>
</list></p>
<p>The Stage 1 generator network uses this combined input to create a low-resolution image (e.g., 64 &#x00D7; 64 pixels) that reflects the basic elements from the text description. This is like a rough sketch capturing the initial form.
<list list-type="bullet">
<list-item>
<p><bold>Training the System</bold></p></list-item>
</list></p>
<p>During this stage, the generator learns to translate basic textual information into simple shapes and structures within the image [<xref ref-type="bibr" rid="ref-47">47</xref>]. Simultaneously, a separate network, the discriminator, trains itself to distinguish real images from these initial attempts.</p>
<p><bold>Stage 2: Refining the Details</bold></p>
<p><list list-type="bullet">
<list-item>
<p><bold>Building on the Foundation</bold></p></list-item>
</list></p>
<p>The next stage combines the low-resolution image from Stage 1 with the original text encoding. This provides the Stage 2 network with both the initial sketch and the textual details.
<list list-type="bullet">
<list-item>
<p><bold>High-Resolution Masterpiece</bold></p></list-item>
</list></p>
<p>The Stage 2 generator network utilizes this information to create a high-resolution image (e.g., 256 &#x00D7; 256 pixels or higher) that closely resembles the described scene.
<list list-type="bullet">
<list-item>
<p><bold>Continuous Improvement</bold></p></list-item>
</list></p>
<p>Similar to Stage 1, both the generator and discriminator in Stage 2 undergo adversarial training [<xref ref-type="bibr" rid="ref-48">48</xref>]. The generator strives to produce increasingly realistic images, while the discriminator refines its ability to discern reality from generated images. In this way, the StackGAN culminates in a high-resolution image that visually aligns with the textual description. It effectively translates the words into a corresponding visual representation.</p>
</sec>
<sec id="s3_3">
<label>3.3</label>
<title>Transformer Models</title>
<p>Transformer models were first introduced by Vaswani et al. [<xref ref-type="bibr" rid="ref-49">49</xref>] in the year 2017. The model is mostly utilized for sophisticated natural language processing applications. OpenAI employed transformers to develop its well-known GPT-2 and GPT-3 [<xref ref-type="bibr" rid="ref-50">50</xref>] models. The transformer architecture has developed and branched out into several forms since its introduction in 2017, going beyond language problems to other domains. They have been applied to forecast time series. They are the main technological advancement underpinning DeepMind&#x2019;s protein structure prediction model, Alpha Fold [<xref ref-type="bibr" rid="ref-51">51</xref>]. Transformers serve as the foundation for Codex, an OpenAI source code creation paradigm. Transformers have more recently made their way into the field of computer vision, where they are gradually taking the place of convolutional neural networks (CNN) in a variety of challenging applications; this is attributed to the attention mechanism in transformers capable of memorizing relative positions of features in images [<xref ref-type="bibr" rid="ref-52">52</xref>]. Transformers can still be improved, and researchers are continuously investigating new uses for them.</p>
<p>OpenAI recently unveiled a unique deep neural network that acquires visual concepts through natural language guidance in January 2021. The Contrastive Language Image Pretraining (CLIP) [<xref ref-type="bibr" rid="ref-53">53</xref>] consists of two encoders, one for images and one for text. CLIP&#x2019;s encoders produce comparable embeddings for images and words, capturing similar concepts. When applied to visual classification tasks without training, CLIP can differentiate between objects X and Y in an image dataset by assessing whether the text description &#x201C;a photo of X&#x201D; or &#x201C;a photo of Y&#x201D; is more likely to be associated with each image.</p>
<p>This transformer has been extensively used in image captioning applications [<xref ref-type="bibr" rid="ref-54">54</xref>,<xref ref-type="bibr" rid="ref-55">55</xref>] due to its ability to create shared representations for image and text prompts. The strong correlation between visual and textual features makes CLIP a highly practical approach. The model proposed by the authors generates an image whose CLIP embedding is most similar to the given text embedding. Through exploration by a genetic algorithm, the generative network produces an optical image.</p>
</sec>
</sec>
<sec id="s4">
<label>4</label>
<title>Generating Images and Videos Using VQGANs</title>
<p>The VQGAN was previously utilized by Ge et al. [<xref ref-type="bibr" rid="ref-15">15</xref>], who generated long-form video using a time-sensitive transformer architecture built on top of 3DVQGAN. This engine was trained on 16-frame clips from benchmark video datasets and effectively produced high-quality videos. The authors used a time-agnostic GAN and encoded temporal information using padding methods between video frames.</p>
<p>Video generation using GANs is not a new concept. Originally designed for image synthesis [<xref ref-type="bibr" rid="ref-37">37</xref>], adapting these methods to video synthesis necessitated encoding the temporal dimension. Some papers have attempted this using RNNs; for instance, MoCoGAN [<xref ref-type="bibr" rid="ref-6">6</xref>] sampled motion vectors to create multiple video frames. More recent methods like MoCoGAN-HD [<xref ref-type="bibr" rid="ref-6">6</xref>] utilize Long Short-Term Memory (LSTM) to assign probabilities to a curve in the abstract multi-dimensional space of a trained picture generator. <xref ref-type="table" rid="table-1">Tables 1</xref> and <xref ref-type="table" rid="table-2">2</xref> highlight the summary of various text-to-image synthesis methods.</p>
<table-wrap id="table-1">
<label>Table 1</label>
<caption>
<title>Summary of text-to-image synthesis methods</title>
</caption>
<table>
<colgroup>
<col align="center" width="20mm"/>
<col align="center" width="20mm"/>
<col align="center" width="55mm"/>
<col align="center" width="55mm"/>
</colgroup>
<thead>
<tr>
<th align="center">Name</th>
<th align="center">Author</th>
<th align="center">Topic</th>
<th align="center">Findings</th>
</tr>
</thead>
<tbody>
<tr>
<td>StackGAN</td>
<td>Zhang et al. [<xref ref-type="bibr" rid="ref-6">6</xref>]</td>
<td>StackGAN: Text to Photorealistic Image Synthesis with Stacked Generative Adversarial Networks</td>
<td>The suggested approach breaks down text-to-image synthesis into a novel process of sketch refinement. Compared to current generative models, this technique produces images with higher quality, more photorealistic features, and greater variety</td>
</tr>
<tr>
<td>AttnGAN</td>
<td>Xu et al. [<xref ref-type="bibr" rid="ref-51">51</xref>]</td>
<td>AttnGAN: FineGrained Text to Image Generation with Attentional Generative Adversarial Networks</td>
<td>This study suggests a new multistage attentional generating network for the AttnGAN to produce high-quality images. Additionally, the authors suggest a deep attentional multimodal similarity model to compute the fine-grained image-text matching loss for training the AttnGAN generator</td>
</tr>
<tr>
<td>StackGAN&#x002B;&#x002B;</td>
<td>Zhang et al. [<xref ref-type="bibr" rid="ref-6">6</xref>]</td>
<td>StackGAN&#x002B;&#x002B;: Realistic Image Synthesis with Stacked Generative Adversarial Networks</td>
<td>This paper talks about two versions of Stack GANs, StackGAN-v1 and StackGAN v2. While StackGANv1 successfully generates 256 &#x00D7; 256 resolution images from text, StackGAN-v2 approximates multiscale image distributions and combines conditional and unconditional image distributions jointly.</td>
</tr>
<tr>
<td>DM-GAN</td>
<td>Zhu et al. [<xref ref-type="bibr" rid="ref-52">52</xref>]</td>
<td>DM-GAN: Dynamic Memory Generative Adversarial Networks for Text-to-Image Synthesis</td>
<td>The proposed methodology introduces the dynamic memory module for handling ill-defined images. DM-GAN outperforms the state-of-the-art models in terms of both qualitative and quantitative metrics.</td>
</tr>
</tbody>
</table>
</table-wrap><table-wrap id="table-2">
<label>Table 2</label>
<caption>
<title>Summary of text-to-image synthesis methods</title>
</caption>
<table>
<colgroup>
<col/>
<col/>
<col/>
</colgroup>
<thead>
<tr>
<th>Model Type</th>
<th>Strengths</th>
<th>Challenges</th>
</tr>
</thead>
<tbody>
<tr>
<td>Diffusion models</td>
<td>High-quality, realistic outputs</td>
<td>Computationally intensive</td>
</tr>
<tr>
<td>GANs</td>
<td>Creative and artistic outputs</td>
<td>Instability during training</td>
</tr>
<tr>
<td>Auto-regressive</td>
<td>Pixel-level accuracy</td>
<td>Slow generation process</td>
</tr>
<tr>
<td>Retrieval-augmented</td>
<td>Enhanced realism with references</td>
<td>Requires robust retrieval mechanisms</td>
</tr>
<tr>
<td>Scene graph-Based</td>
<td>Structured and relationship-aware</td>
<td>Complex preprocessing</td>
</tr>
<tr>
<td>Multimodal transformers</td>
<td>Unified learning of text and image</td>
<td>Requires large-scale training data</td>
</tr>
</tbody>
</table>
</table-wrap>
<p>It&#x2019;s worth noting that the CLIP transformer has historically served as a guidance tool for manipulating images [<xref ref-type="bibr" rid="ref-55">55</xref>]. Recently, Vector Quantized Variational Autoencoder (VQ-VAE) based visual models for video synthesis tasks have been introduced, involving transforming pictures into discrete tokens or frames. This technology has the potential to provide extensive pretraining for text-to image creation. However, in the problem statement addressed in this paper, the authors have implemented a speech-to-image generator. Speech-to-image translation is an extremely complex task involving the creation of a speech embedding network and generating paired probabilities of parts-of-spoken-word-to-image pairs [<xref ref-type="bibr" rid="ref-40">40</xref>].</p>
<p>Generating images from speech input introduces unique challenges due to the fundamental differences between auditory and visual modalities. Unlike text, which has clear symbolic representations, speech involves nuances like pitch, tone, rhythm, and phonetic variability that must be mapped accurately to visual information. Using speech input for image generation introduces unique challenges that span speech recognition, natural language understanding, and image synthesis. Speech-to-Text Conversion Accuracy-Errors in automatic speech recognition (ASR) can increase incorrect textual inputs, that leads to incorrect image generation. Its solution can be training on diverse and domain-specific datasets to improve recognition accuracy. Spoken descriptions are often less structured and more ambiguous than written text. Also, variations in tone, accent, pacing, and phrasing can affect the system&#x2019;s interpretation. A possible solution would be to leverage large-scale language models (e.g., GPT) to parse and refine ambiguous speech into structured textual descriptions. Generating images from live speech inputs requires low latency for applications like interactive tools as it should be generated in real-time. Using lightweight, optimized models for ASR and image synthesis (e.g., efficient diffusion models) can make real-time processing possible. Speech systems often struggle with accents, dialects, and noisy environments, affecting downstream tasks like image generation. Training ASR models with diverse datasets that include different accents and dialects will make the model robust. This paper has devised a method to simplify the process of creating complex embeddings. It implements a clever sequence of operations to translate speech into tokenized words and utilizes pre-trained models in conjunction to generate beautiful imagery from speech expressions.</p>
</sec>
<sec id="s5">
<label>5</label>
<title>Proposed Approach</title>
<p>This study investigates a new modality in text-to-image generation by using speech prompts as inputs, which is novel compared to traditional text prompts. The use of speech allows for more adaptable and natural interactions, which helps applications in accessibility, virtual assistants, and interactive media. The approach is also useful for the users with visual impairments or those in hands-free environments. The model uses CLIP&#x2019;s pre-trained language-image embeddings with a cosine similarity loss function to calculate and optimize the resemblance between the encoded text and image representations. This approach guarantees that generated images align closely with the semantic content of the speech prompt. Using cosine similarity for alignment enhances the interpretability and accuracy of generated images, addressing common issues in generative models where images may only loosely match prompts. Optimizing the latent space, which is critical, allows for finer control over the generated content, enhancing the clarity and fidelity of the final images. The architecture merges CLIP for text-image alignment, a GAN for high-quality image synthesis, and a speech-to-text component to translate spoken prompts into text. This tri-component strategy represents a novel pipeline for multimodal image generation, leveraging the strengths of each model. This approach enables real-time, multimodal interaction where spoken words can directly drive the creation of images. The combination of CLIP and VQGAN facilitates natural language understanding and high-quality image synthesis abilities. The system begins with a random noise or an initial latent vector (image representation), where VQGAN generates an image iteratively. CLIP evaluates the alignment of the generated image with the textual description provided. The system optimizes the generated image by adjusting the latent space in VQGAN, which is driven by CLIP&#x2019;s feedback loop. The study explains using CLIP, highlighting its capacity to generalize across visual domains with minimal training, similar to GPT models. The deployment of CLIP&#x2019;s pre-trained strengths decreases the need for comprehensive dataset-specific training, making the model more efficient and versatile for new applications. The speech input with CLIP, GAN, and iterative latent space optimization, generates better quality and highly realistic images.</p>
<p>CLIP is designed to combine visual and textual understanding in a single model as shown in <xref ref-type="fig" rid="fig-3">Fig. 3</xref>. Its primary function is understanding and retrieving images based on natural language depictions without requiring specialized training datasets for each task. It utilizes a contrastive learning approach, learning from an extensive dataset of images and their captions. CLIP learns to map images and text into a shared embedding space where similar images and captions are close. CLIP&#x2019;s capability to correspond text to images has made it widely applicable for various AI tasks, including image search, text-based image manipulation, and guiding other generative models (like DALL-E) to produce images aligned with textual descriptions. StackGAN&#x002B;&#x002B; is useful for text-to-image generation tasks, mainly where the model generates images that closely match the input textual description, with more definitive colors and sensible textures than previous GAN-based models. Combining CLIP and StackGAN&#x002B;&#x002B; entitles better control over text-to-image generation. CLIP can act as a &#x201C;guiding model&#x201D; for StackGAN&#x002B;&#x002B;, estimating and leading the image generation toward nearer alignment with the input text description. By employing CLIP&#x2019;s capability to diagnose text instructions, it is possible to use a StackGAN&#x002B;&#x002B;-like model to refine images iteratively, considering the user feedback. While CLIP is good at aligning images and text, StackGAN&#x002B;&#x002B; concentrates on forming intricate images from descriptions; combined, they create a robust technique for high-quality, interactive image generation. CLIP&#x2019;s semantic understanding to maintain relevance to the text is its highlight. CLIP&#x2019;s robust semantic understanding can mitigate minor ASR errors in speech-to-text conversion, ensuring the generated images remain relevant. VQGAN &#x002B; CLIP can handle abstract or vague descriptions often found in spoken language. It also supports real-time applications like interactive art tools. Traditional GANs have limited semantic text-image matching but CLIP &#x002B; VQGAN can easily generalize the match and produce high quality images. The highlight of our model is it works with unpaired data in the unsupervised learning environment unlike other traditional models. The working mechanism of CLIP &#x002B; VQGAN begins with the user providing a text prompt, such as &#x201C;A dragon.&#x201D; VQGAN initializes with a random latent vector z, representing an initial image in its latent space. The text prompt is encoded into a text embedding using CLIP&#x2019;s text encoder, while the image generated from z is encoded by CLIP&#x2019;s image encoder to obtain an image embedding. The cosine similarity between these embeddings serves as a score for how well the image matches the prompt. Through an optimization loop, z is updated via gradient descent to maximize this similarity, with regularization terms ensuring image coherence and quality. The process iterates until the image aligns with the text prompt, and the optimized z is decoded by VQGAN to produce the final image. Thus, combining CLIP for the multimodal understanding with the generative capabilities of VQGAN works very well for textual concepts and visual imagination. Some of the challenges for speech recognition systems include pronunciations, lingoes, background noise, or vague pronunciation, leading to inaccuracies in the transcribed text. For example, Homophones (e.g., &#x201C;flower&#x201D; vs. &#x201C;flour&#x201D;) can be misinterpreted, especially if the context is ambiguous. Another challenge is that the proper nouns, technical jargon, or cultural idioms may need to be recognized correctly during this conversion. Also, spoken language requires more structure than written language (e.g., fillers, pauses, or fragmented sentences). Sometimes, even accurate transcription can deliver incomplete or unclear descriptions (e.g., &#x201C;a large animal&#x201D; could mean any large animal, like an elephant, a whale, or a horse).</p>
<fig id="fig-3">
<label>Figure 3</label>
<caption>
<title>Functioning of stack-generative adversarial networks</title>
</caption>
<graphic mimetype="image" mime-subtype="tif" xlink:href="CMES_58456-fig-3.tif"/>
</fig>
<p>Speech usually requires the inclusion of details or prompts necessary for explicit image generation. Speech-based contextual cues like tone or intonation may be misinterpreted during conversion, potentially losing nuances like emphasis or implied meanings. The Speech may rely on situational or cultural context to convey meaning (e.g., &#x201C;a ceremonial dress&#x201D; or &#x201C;traditional dress&#x201D; can convey vastly different things based on the territory). Image generation models may need clear, detailed information. The Contrastive Language-Image Pretraining (CLIP) model is a fundamental component in the proposed architecture.</p>
<p>The (CLIP) model aligns image embeddings and text embeddings in a shared latent space. This is integration achieved through a contrastive loss function. The contrastive loss function guarantees that matching text-image pairs are more similar in the latent space than mismatched pairs. The CLIP loss is computed as the negative logarithm of a fraction. The numerator of this fraction is the exponentiated similarity score between the text embedding &#x03C4;&#x03C4; and the image embedding II, divided by the temperature parameter &#x03C4;&#x03C4;. Compared to non-matching pairs, the loss shows the high similarity between matching text-image pairs, optimizing the model to align text and image embeddings in the shared space.</p>
<p>CLIP model aligns image embeddings and text embeddings in a shared latent space. The contrastive loss function ensures that matching text-image pairs are more similar in the latent space than mismatched pairs. The CLIP loss is calculated as the negative logarithm of a fraction. The numerator of this fraction is the exponentiated similarity score between the text embedding TT and the image embedding II, divided by the temperature parameter &#x03C4;&#x03C4;. Compared to non-matching pairs, the loss indicates the high similarity between matching text-image pairs, optimizing the model to align text and image embeddings in the shared space. The contrastive loss function is mathematically stated as shown in <xref ref-type="disp-formula" rid="eqn-1">Eq. (1)</xref>.
<disp-formula id="eqn-1"><label>(1)</label><mml:math id="mml-eqn-1" display="block"><mml:mrow><mml:mtext>L</mml:mtext></mml:mrow><mml:mi mathvariant="normal">&#x005F;</mml:mi><mml:mrow><mml:mtext>CLIP&#xA0;</mml:mtext></mml:mrow><mml:mo>=</mml:mo><mml:mo>&#x2212;</mml:mo><mml:mi>log</mml:mi><mml:mo>&#x2061;</mml:mo><mml:mo>&#x2211;</mml:mo><mml:msup><mml:mrow><mml:mtext>I</mml:mtext></mml:mrow><mml:mrow><mml:mi mathvariant="normal">&#x2032;</mml:mi></mml:mrow></mml:msup><mml:mi>exp</mml:mi><mml:mo>&#x2061;</mml:mo><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mtext>sim</mml:mtext></mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mtext>T</mml:mtext></mml:mrow><mml:mo>,</mml:mo><mml:msup><mml:mrow><mml:mtext>I</mml:mtext></mml:mrow><mml:mrow><mml:mi mathvariant="normal">&#x2032;</mml:mi></mml:mrow></mml:msup><mml:mo stretchy="false">)</mml:mo><mml:mrow><mml:mo>/</mml:mo></mml:mrow><mml:mrow><mml:mi mathvariant="normal">&#x03C4;</mml:mi></mml:mrow><mml:mo stretchy="false">)</mml:mo><mml:mi>exp</mml:mi><mml:mo>&#x2061;</mml:mo><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mtext>sim</mml:mtext></mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mtext>T</mml:mtext></mml:mrow><mml:mo>,</mml:mo><mml:mrow><mml:mtext>I</mml:mtext></mml:mrow><mml:mo stretchy="false">)</mml:mo><mml:mrow><mml:mo>/</mml:mo></mml:mrow><mml:mrow><mml:mi mathvariant="normal">&#x03C4;</mml:mi></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:math></disp-formula>where:</p>
<p><bold>Similarity Function (sim(T, I) sim(T, I)):</bold> Measures the similarity between the text embedding TT and the image embedding II. Similarity Function is typically implemented as the dot product of normalized embeddings.</p>
<p><bold>Temperature Parameter (&#x03C4;&#x03C4;):</bold> The Temperature Parameter is a learnable or fixed scalar value that controls the sharpness of the probability distribution.</p>
<p>The numerator represents the similarity between the text TT and its corresponding image II, scaled by the temperature parameter. The denominator states the summation over all possible image embeddings <inline-formula id="ieqn-1"><mml:math id="mml-ieqn-1"><mml:msup><mml:mrow><mml:mi mathvariant="normal">I</mml:mi></mml:mrow><mml:mrow><mml:mi mathvariant="normal">&#x2032;</mml:mi></mml:mrow></mml:msup><mml:msup><mml:mrow><mml:mi mathvariant="normal">I</mml:mi></mml:mrow><mml:mrow><mml:mi mathvariant="normal">&#x2032;</mml:mi></mml:mrow></mml:msup></mml:math></inline-formula>, normalizing the probability.</p>
<p><bold>Logarithm and Negative Sign:</bold> The negative log ensures that the loss decreases as the similarity of the correct text-image pair increases relative to other pairs.</p>
<p>This loss function trains the model to maximize the similarity of matching text-image pairs while minimizing the similarity of non-matching pairs.</p>
<p><bold>Dual-Path Contrastive Loss:</bold> In practice, the contrastive loss is computed for both the Text-to-Image Alignment, which is Matching a text embedding TT with its correct image embedding II, and the Image-to-Text Alignment, matching an image embedding II with its corresponding text embedding TT.</p>
<p>The VQGAN architecture consists of an Encoder, which maps input images x to a latent representation ze(x). It has a Quantizer where Vector quantization is applied to ze(x), mapping it to the closest discrete vector zq in a learned codebook. Further, the Decoder Decodes zq back into the reconstructed image G(zq). Conversely, the Discriminator guides the generator via adversarial training, making reconstructed images visually realistic. The Vector Quantized Generative Adversarial Network (VQGAN) integrates adversarial training with vector quantization and a reconstruction objective to learn discrete latent representations.</p>
<p>The total loss function for the VQGAN is given by:
<disp-formula id="eqn-2"><label>(2)</label><mml:math id="mml-eqn-2" display="block"><mml:mrow><mml:mtext>LVQGAN&#xA0;</mml:mtext></mml:mrow><mml:mo>=</mml:mo><mml:mrow><mml:mtext>&#xA0;LGAN&#xA0;</mml:mtext></mml:mrow><mml:mo>+</mml:mo><mml:mrow><mml:mtext>&#xA0;Lrecon&#xA0;</mml:mtext></mml:mrow><mml:mo>+</mml:mo><mml:mrow><mml:mtext>&#xA0;Lcommit</mml:mtext></mml:mrow></mml:math></disp-formula></p>
<p>The Loss Components for VQGAN include:</p>
<p><bold>Adversarial Loss (LGAN):</bold> This loss is observed from the adversarial training framework, where the generator aims to generate pragmatic images, and the discriminator differentiates between real and generated images. Mathematically, it is expressed as given below.
<disp-formula id="eqn-3"><label>(3)</label><mml:math id="mml-eqn-3" display="block"><mml:mrow><mml:mtext>LGAN&#xA0;</mml:mtext></mml:mrow><mml:mo>=</mml:mo><mml:mrow><mml:mtext>E</mml:mtext></mml:mrow><mml:mo stretchy="false">[</mml:mo><mml:mrow><mml:mtext>logD</mml:mtext></mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mtext>x</mml:mtext></mml:mrow><mml:mo stretchy="false">)</mml:mo><mml:mo stretchy="false">]</mml:mo><mml:mo>+</mml:mo><mml:mrow><mml:mtext>E</mml:mtext></mml:mrow><mml:mo stretchy="false">[</mml:mo><mml:mi>log</mml:mi><mml:mo>&#x2061;</mml:mo><mml:mo stretchy="false">(</mml:mo><mml:mn>1</mml:mn><mml:mo>&#x2212;</mml:mo><mml:mrow><mml:mtext>D</mml:mtext></mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mtext>G</mml:mtext></mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mtext>z</mml:mtext></mml:mrow><mml:mo stretchy="false">)</mml:mo><mml:mo stretchy="false">)</mml:mo><mml:mo stretchy="false">)</mml:mo><mml:mo stretchy="false">]</mml:mo></mml:math></disp-formula>where:</p>
<p>D(x): Output of the discriminator for a real image xx.</p>
<p>G(z)G(z): Generated image from latent code z by the generator.</p>
<p><bold>Reconstruction Loss (Lrecon):</bold> Encourages the generator to produce images that closely resemble the original inputs after encoding and decoding through the latent space. It is commonly implemented as an &#x2113;1\ell_1&#x2113;1 or &#x2113;2\ell_2&#x2113;2 norm:
<disp-formula id="eqn-4"><label>(4)</label><mml:math id="mml-eqn-4" display="block"><mml:mrow><mml:mtext>Lrecon&#xA0;</mml:mtext></mml:mrow><mml:mo>=&#x2225;</mml:mo><mml:mrow><mml:mtext>x</mml:mtext></mml:mrow><mml:mo>&#x2212;</mml:mo><mml:mrow><mml:mtext>G</mml:mtext></mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mtext>E</mml:mtext></mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mtext>x</mml:mtext></mml:mrow><mml:mo stretchy="false">)</mml:mo><mml:mo stretchy="false">)</mml:mo><mml:mo>&#x2225;</mml:mo><mml:mrow><mml:mtext>p</mml:mtext></mml:mrow></mml:math></disp-formula>where:</p>
<p>E(x): Encoder output (latent representation).</p>
<p>p: Typically, 11 (MAE) or 22 (MSE).</p>
<p><bold>Commitment Loss (Lcommit):</bold> Ensures that the Encoder efficiently uses the learned discrete codebook by penalizing large deviations between the encoder output and its nearest codebook vector.
<disp-formula id="eqn-5"><label>(5)</label><mml:math id="mml-eqn-5" display="block"><mml:mrow><mml:mtext>Lcommit&#xA0;</mml:mtext></mml:mrow><mml:mo>=</mml:mo><mml:mrow><mml:mi mathvariant="normal">&#x03B2;</mml:mi></mml:mrow><mml:mo>&#x2225;</mml:mo><mml:mrow><mml:mtext>ze&#xA0;</mml:mtext></mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mtext>x</mml:mtext></mml:mrow><mml:mo stretchy="false">)</mml:mo><mml:mo>&#x2212;</mml:mo><mml:mrow><mml:mtext>&#xA0;sg</mml:mtext></mml:mrow><mml:mo stretchy="false">[</mml:mo><mml:mrow><mml:mtext>zq</mml:mtext></mml:mrow><mml:mo stretchy="false">]</mml:mo><mml:mo>&#x2225;</mml:mo><mml:mn>22</mml:mn></mml:math></disp-formula>where:</p>
<p>ze(x): Latent representation output by the Encoder.</p>
<p>zq: Closest vector from the discrete codebook.</p>
<p>sg[&#x22C5;]: Stop gradient operation to prevent gradients from flowing into the codebook during optimization.</p>
<p>&#x03B2;: Hyperparameter controlling the weight of the commitment loss.</p>
<p><bold>Fr&#x00E9;chet Inception Distance (FID):</bold> The evaluation of the model was conducted using alternative methods as Fr&#x00E9;chet Inception Distance (FID). FID is used in evaluating generative models, particularly in the context of image synthesis. FID is very popular because it quantifies both the quality and diversity of generated images in a single score and both quality and diversity are the most important evaluation measures of GANs. Other pixel-wise metrics such as Mean Squared Error (MSE) or Structural Similarity Index (SSIM), which are often too rigid to capture perceptual quality, FID compares the distributions of real and generated images in a feature space. The characteristic of FID is its alignment with human perception of the generated image quality. It captures high-level semantic features. It evaluates the entire distribution of generated images, making it sensitive to mode collapse. Inception Score is also used in measuring image quality but for our dataset there are biasing issues because ImageNet is a distinct and well-defined categories dataset. Fr&#x00E9;chet Inception Distance (FID) is stated as shown in <xref ref-type="disp-formula" rid="eqn-6">Eq. (6)</xref>.
<disp-formula id="eqn-6"><label>(6)</label><mml:math id="mml-eqn-6" display="block"><mml:mrow><mml:mtext>FID</mml:mtext></mml:mrow><mml:mo>=&#x2225;</mml:mo><mml:mrow><mml:mi mathvariant="normal">&#x0B5;</mml:mi></mml:mrow><mml:mn>1</mml:mn><mml:mo>&#x2212;</mml:mo><mml:mrow><mml:mi mathvariant="normal">&#x0B5;</mml:mi></mml:mrow><mml:mn>2</mml:mn><mml:mo>&#x2225;</mml:mo><mml:mn>2</mml:mn><mml:mo>+</mml:mo><mml:mrow><mml:mtext>&#xA0;Tr</mml:mtext></mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mi mathvariant="normal">&#x03A3;</mml:mi></mml:mrow><mml:mn>1</mml:mn><mml:mo>+</mml:mo><mml:mrow><mml:mi mathvariant="normal">&#x03A3;</mml:mi></mml:mrow><mml:mn>2</mml:mn><mml:mo>&#x2212;</mml:mo><mml:mn>2</mml:mn><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mi mathvariant="normal">&#x03A3;</mml:mi></mml:mrow><mml:mn>1</mml:mn><mml:mrow><mml:mi mathvariant="normal">&#x03A3;</mml:mi></mml:mrow><mml:mn>2</mml:mn><mml:mo stretchy="false">)</mml:mo><mml:mn>1</mml:mn><mml:mrow><mml:mo>/</mml:mo></mml:mrow><mml:mn>2</mml:mn><mml:mo stretchy="false">)</mml:mo></mml:math></disp-formula>where &#x03BC;1 and &#x03BC;2 are the mean feature vectors of the generated and real images, and <inline-formula id="ieqn-2"><mml:math id="mml-ieqn-2"><mml:mrow><mml:mi mathvariant="normal">&#x03A3;</mml:mi></mml:mrow></mml:math></inline-formula>1 and <inline-formula id="ieqn-3"><mml:math id="mml-ieqn-3"><mml:mrow><mml:mi mathvariant="normal">&#x03A3;</mml:mi></mml:mrow></mml:math></inline-formula>2 are their covariance matrices.</p>
<p>Inception Score (IS) which is another evaluation parameter, can not be applied here because of the ImageNet dataset characteristics. However, we have shown best and worst results achieved through our models in <xref ref-type="table" rid="table-3">Table 3</xref>. In addition to the results shown in <xref ref-type="table" rid="table-3">Table 3</xref>, we have also compared our model results with other two SOTA approaches of image generation StackGAN&#x002B;&#x002B; and Dall-E and represented results in <xref ref-type="table" rid="table-4">Table 4</xref>. Since this is a text quality dependent model, its limitations related to text to speech generation are also mentioned in the discussion section. Human inspection is also required to validate the GANs results so the generated images were validated by our team and some of the students who work in the similar domain. Our model has the potential of generating high quality realistic images in various categories and shows its robustness.</p>
<table-wrap id="table-3">
<label>Table 3</label>
<caption>
<title>FID score for generated images</title>
</caption>
<table>
<colgroup>
<col/>
<col/>
<col/>
<col/>
</colgroup>
<thead>
<tr>
<th>Prompt</th>
<th>Generated image</th>
<th>Real image</th>
<th>FID score</th>
</tr>
</thead>
<tbody>
<tr>
<td>Komodo dragon</td>
<td><inline-graphic mime-subtype="tif" xlink:href="CMES_58456-inline-1.tif"/></td>
<td><inline-graphic mime-subtype="tif" xlink:href="CMES_58456-inline-2.tif"/></td>
<td>31.250</td>
</tr>
<tr>
<td>Tennis ball</td>
<td><inline-graphic mime-subtype="tif" xlink:href="CMES_58456-inline-3.tif"/></td>
<td><inline-graphic mime-subtype="tif" xlink:href="CMES_58456-inline-4.tif"/></td>
<td>28.750</td>
</tr>
<tr>
<td>Tarantula</td>
<td><inline-graphic mime-subtype="tif" xlink:href="CMES_58456-inline-5.tif"/></td>
<td><inline-graphic mime-subtype="tif" xlink:href="CMES_58456-inline-6.tif"/></td>
<td>35.125</td>
</tr>
<tr>
<td>Teddy bear</td>
<td><inline-graphic mime-subtype="tif" xlink:href="CMES_58456-inline-7.tif"/></td>
<td><inline-graphic mime-subtype="tif" xlink:href="CMES_58456-inline-8.tif"/></td>
<td>40.375</td>
</tr>
<tr>
<td>Bear</td>
<td><inline-graphic mime-subtype="tif" xlink:href="CMES_58456-inline-9.tif"/></td>
<td><inline-graphic mime-subtype="tif" xlink:href="CMES_58456-inline-10.tif"/></td>
<td>42.76</td>
</tr>
<tr>
<td>A man</td>
<td><inline-graphic mime-subtype="tif" xlink:href="CMES_58456-inline-11.tif"/></td>
<td><inline-graphic mime-subtype="tif" xlink:href="CMES_58456-inline-12.tif"/></td>
<td>13.24</td>
</tr>
<tr>
<td>Shaktimaan-superhero</td>
<td><inline-graphic mime-subtype="tif" xlink:href="CMES_58456-inline-13.tif"/></td>
<td><inline-graphic mime-subtype="tif" xlink:href="CMES_58456-inline-14.tif"/></td>
<td>17.12</td>
</tr>
</tbody>
</table>
</table-wrap><table-wrap id="table-4">
<label>Table 4</label>
<caption>
<title>Comparison of proposed model with existing models</title>
</caption>
<table>
<colgroup>
<col/>
<col/>
<col/>
<col/>
</colgroup>
<thead>
<tr>
<th>Prompt</th>
<th>CLIP &#x002B; VQGAN</th>
<th>StackGAN&#x002B;&#x002B;</th>
<th>Dall-E</th>
</tr>
</thead>
<tbody>
<tr>
<td>Shaktimaan-superman</td>
<td><inline-graphic mime-subtype="tif" xlink:href="CMES_58456-inline-15.tif"/></td>
<td><inline-graphic mime-subtype="tif" xlink:href="CMES_58456-inline-16.tif"/></td>
<td><inline-graphic mime-subtype="tif" xlink:href="CMES_58456-inline-17.tif"/></td>
</tr>
<tr>
<td>Teddy bear</td>
<td><inline-graphic mime-subtype="tif" xlink:href="CMES_58456-inline-18.tif"/></td>
<td><inline-graphic mime-subtype="tif" xlink:href="CMES_58456-inline-19.tif"/></td>
<td><inline-graphic mime-subtype="tif" xlink:href="CMES_58456-inline-20.tif"/></td>
</tr>
<tr>
<td>Bear</td>
<td><inline-graphic mime-subtype="tif" xlink:href="CMES_58456-inline-21.tif"/></td>
<td><inline-graphic mime-subtype="tif" xlink:href="CMES_58456-inline-22.tif"/></td>
<td><inline-graphic mime-subtype="tif" xlink:href="CMES_58456-inline-23.tif"/></td>
</tr>
<tr>
<td>Komodo dragon</td>
<td><inline-graphic mime-subtype="tif" xlink:href="CMES_58456-inline-24.tif"/></td>
<td><inline-graphic mime-subtype="tif" xlink:href="CMES_58456-inline-25.tif"/></td>
<td><inline-graphic mime-subtype="tif" xlink:href="CMES_58456-inline-26.tif"/></td>
</tr>
<tr>
<td>Tarantula</td>
<td><inline-graphic mime-subtype="tif" xlink:href="CMES_58456-inline-27.tif"/></td>
<td><inline-graphic mime-subtype="tif" xlink:href="CMES_58456-inline-28.tif"/></td>
<td><inline-graphic mime-subtype="tif" xlink:href="CMES_58456-inline-29.tif"/></td>
</tr>
<tr>
<td>Tennis ball</td>
<td><inline-graphic mime-subtype="tif" xlink:href="CMES_58456-inline-30.tif"/></td>
<td><inline-graphic mime-subtype="tif" xlink:href="CMES_58456-inline-31.tif"/></td>
<td><inline-graphic mime-subtype="tif" xlink:href="CMES_58456-inline-32.tif"/></td>
</tr>
<tr>
<td>A Man</td>
<td><inline-graphic mime-subtype="tif" xlink:href="CMES_58456-inline-33.tif"/></td>
<td><inline-graphic mime-subtype="tif" xlink:href="CMES_58456-inline-34.tif"/></td>
<td><inline-graphic mime-subtype="tif" xlink:href="CMES_58456-inline-35.tif"/></td>
</tr>
</tbody>
</table>
</table-wrap>
<p>The dataset used for this particular experiment is the ImageNet dataset, first introduced in 2009 [<xref ref-type="bibr" rid="ref-19">19</xref>]. This dataset is built on the WordNet structure and contains over 30 million annotated images, serving as benchmark standards for numerous computer vision applications. The implementation of VQGAN used by the authors of this paper learns from a codebook that encodes images as tokens. In this version, a reduction actor of 16 is utilized. Therefore, an image of size 256 &#x00D7; 256 would be normalized to 16 &#x00D7; 16 as shown in <xref ref-type="fig" rid="fig-4">Fig. 4</xref>.</p>
<fig id="fig-4">
<label>Figure 4</label>
<caption>
<title>Sample images tested in this research [<xref ref-type="bibr" rid="ref-19">19</xref>]</title>
</caption>
<graphic mimetype="image" mime-subtype="tif" xlink:href="CMES_58456-fig-4.tif"/>
</fig>
<p>CLIP&#x2019;s robust semantic understanding can mitigate minor ASR errors in speech-to-text conversion, ensuring the generated images remain relevant. VQGAN &#x002B; CLIP can handle abstract or vague descriptions often found in spoken language. It also supports real-time applications like interactive art tools. Traditional GANs have limited semantic text-image matching but CLIP&#x002B;VQGAN can easily generalize the match and produce high quality images. The highlight of our model is it works with unpaired data in the unsupervised learning environment unlike other traditional models. DALL-E is computationally intensive. It also has less control over style compared to other techniques like VQGAN &#x002B; CLIP or diffusion models. Even stable diffusion models have limitations such as slow generation and also, they are resource intensive. Prompt sensitivity is also another problem with diffusion models. The objective of this paper is to take input speech from the user, convert it into text using the Chrome Web-to-Speech API, and then feed this text into our proposed VQGAN &#x002B; CLIP model to generate beautiful imagery as shown in <xref ref-type="fig" rid="fig-3">Fig. 3</xref>.</p>

<p>The authors aim to establish an optimal model for image generation using speech prompts and enhance the quality of the generated image. The text prompt provided to the model will include specific words and phrases that should be included in the resulting image. The model will encode these prompts and utilize a cosine similarity loss function to assess the similarity between the encoded text and the encoded image, ensuring they are indistinguishable. The model will keep on updating its latent space parameters until the desired image is generated. The model is divided into three parts: I. CLIP, II. GAN, III. Speech-to-text translation Contrastive Language Image Pretraining (CLIP) is a neural network model that has already been trained on over 400 million (image, text) pairs. As its name implies, the architecture produces a relevant image for the provided caption to the model. The rationale for selecting CLIP primarily lies in its ability to rapidly learn visual concepts through natural language supervision (<xref ref-type="table" rid="table-2">Table 2</xref>). CLIP can be applied to any visual classification benchmark similar to the zero-shot capabilities of GPT-2 and GPT-3 by merely providing the names of the visual categories to be recognized. The CLIP model is built using the following sub-models, as shown in <xref ref-type="fig" rid="fig-5">Figs. 5</xref> and <xref ref-type="fig" rid="fig-6">6</xref>. Text Encoder and Image Encoder, these sub-models will embed the text and images to a mathematical space, and their dot product is used to check the similarity score. <xref ref-type="fig" rid="fig-5">Fig. 5</xref> shows the basic architecture and <xref ref-type="fig" rid="fig-6">Fig. 6</xref> shows the vision transformer.</p>
<fig id="fig-5">
<label>Figure 5</label>
<caption>
<title>Block diagram of the proposed method</title>
</caption>
<graphic mimetype="image" mime-subtype="tif" xlink:href="CMES_58456-fig-5.tif"/>
</fig><fig id="fig-6">
<label>Figure 6</label>
<caption>
<title>Basic architecture used in CLIP</title>
</caption>
<graphic mimetype="image" mime-subtype="tif" xlink:href="CMES_58456-fig-6.tif"/>
</fig>
<p>The Speech-to-Text model used in this approach is the JavaScript Web Speech Application Programming Interface (API). Its ability to recognize language is affected by a lot of external factors such as speech clarity, browser support, speaker accent, and pronunciation. Variations in speech recognition accuracy can impact the final image generation, potentially leading to less accurate visual representations. To ensure optimal speech transcription, the model uses word processing techniques such as a language model and lexicon to optimally match the processed words to proper keywords to extract the most information from the transcription. This is done via the CLIP transformer as explained in the section proposed approach.</p>
<p>The accuracy of Speech-to-Text translation plays a pivotal role in the quality of the images generated by our model. Inaccuracies in the Speech-to-Text output, such as misinterpretations of unclear speech or background noise, can lead to incorrect or unintended visual representations. For instance, ambiguous or misrecognized words may result in images that do not align with the intended speech input. To mitigate these issues, this model can be integrated with more robust Speech-to-Text systems that are better at handling diverse speech inputs, including accents and noisy environments. Future work may also explore preprocessing techniques to refine the Speech-to-Text output before feeding it into the image generation pipeline, thereby enhancing the system&#x2019;s overall reliability.</p>
</sec>
<sec id="s6">
<label>6</label>
<title>Results and Discussion</title>
<p>The images shown in <xref ref-type="table" rid="table-2">Table 2</xref>, display the results obtained by the Generative Adversarial Networks on the ImageNet dataset. It was noted that the quality of the generated images improves with an increase in the number of iterations. However, it was also noted that an increase in the number of iterations also correlates with an increase in the computational time. The authors explored the effects of iteration count on image quality and computational cost. The images generated from the textual description of &#x201C;Roller-coaster&#x201D; over various iterations have been presented below. The images produced using the proposed method also contain some amount of noise and, thus, are not very sharp due to the large number of classes in the dataset and the various weights that were assigned to each class by the CLIP transformer. Nevertheless, they still incorporate elements from the provided input prompts, illustrating that the model is capable of capturing data from verbal input and generating a corresponding synthesized image despite a low iteration count.</p>

<p>The authors calculated the FID score for the outputs as a quantitative measure to assess the model [<xref ref-type="bibr" rid="ref-50">50</xref>]. The FID score is a popular statistic for evaluating the quality of produced pictures compared to actual photos, measuring how closely the distribution of characteristics in produced pictures resembles that of real photos. However, the FID score is not sufficient to represent the characteristics of a model designed for any newly generated content.</p>
<p>The Inception Score (IS) and FID are very popular evaluation metrics for checking the quality of generated images [<xref ref-type="bibr" rid="ref-51">51</xref>]. The IS shows how well the generated images resemble real images by assessing their diversity. Higher IS values indicate more diverse and realistic images. The FID score, compares the distribution of features extracted from real and generated images, with lower scores indicating closer similarity to real images. These metrics were chosen for their effectiveness in capturing both the quality and diversity of generated images, making them ideal for evaluating the model&#x2019;s performance. It is calculated by first using a pre-trained InceptionV3 network to extract feature vectors from both sets of images. The mean and covariance of these feature vectors are then computed for both the real images and the generated images. The FID score is determined by the Frechet distance formula, which measures the distance between the two multivariate Gaussian distributions represented by these means and covariances.</p>
<p>The model achieved an Inception Score of 72 and an FID score of 28.75, showcasing its capability to produce high-quality and diverse images. In this study, the FID score was relatively low, mainly because the VQGAN f16 artifacts were penalized harshly. Despite the noisy generated images, the authors emphasize that the proposed model&#x2019;s intended purpose is to generate abstract visuals using speech inputs. The outputs of prompts containing descriptions of objects do not present in the ImageNet dataset used by VQGAN, or the zero-shot generated images, demonstrate the diverse nature of the model and its outputs. The model can generate unique, highly imaginative, and abstract images that inspire artistic creativity.</p>
<p>The authors acknowledge that the proposed method has limitations, and the images generated are not perfect due to the large number of classes in the dataset and the various weights assigned to each class by the CLIP transformer. When generating an image from a textual prompt, the GANs and transformers attempt to map the textual input to an image output. However, the mapping is not always straightforward, especially when dealing with abstract concepts or unusual descriptions. As a result, the model sometimes produces noisy and blurry images. However, they believe that their research provides a promising proof-of-concept for future work on the use of speech prompts and GANs for generating imaginative and abstract images. Additionally, the authors suggest extending the method to generate animations and videos using speech prompts, opening up new possibilities for creative expression. The diversity of the dataset introduces some level of noise into the model&#x2019;s understanding of the images, which can lead to the generation of suboptimal images. Despite these limitations, the authors believe that their method demonstrates promising results and opens up new possibilities for creative expression.</p>
<p>Handling the computational demands associated with fine-tuning the CLIP and VQGAN models is challenging. Due to limited processing capacity, it was crucial to optimize the training process to ensure efficient use of resources. Employed techniques such as model pruning and mixed-precision training to reduce the computational load while maintaining the quality of the generated images. Additionally, the use of noise reduction and normalization techniques during preprocessing to handle variations in speech input, such as different accents, speech speeds, and background noises, significantly affects the accuracy of the Speech-to-Text translation.</p>
<p>The generated images contain noise and are not very sharp due to the large number of classes in the dataset and the various weights assigned to each class by the CLIP transformer; they contain elements from the provided input prompts. Therefore, the model can capture information through speech input and generate a coinciding synthesized image despite the low iteration count. However, the intended purpose of the proposed model is to generate abstract visuals using speech inputs, and the outputs of prompts containing descriptions of objects not present in the ImageNet dataset used by the VQGAN exhibit the diverse nature of the model and its outputs shown in <xref ref-type="fig" rid="fig-7">Fig. 7</xref>.</p>
<fig id="fig-7">
<label>Figure 7</label>
<caption>
<title>Roller-coaster generated by CLIP</title>
</caption>
<graphic mimetype="image" mime-subtype="tif" xlink:href="CMES_58456-fig-7.tif"/>
</fig>
<p>The authors have also conducted a quantitative analysis of the model by calculating the FID score for the outputs. However, they believe that a model intended for vague and imaginative image generation can have numerous use cases, and the FID score does not fully capture these capabilities. The FID score was obtained using the open-source implementation of FID Score&#x2014;&#x201C;pytorch-FID&#x201D; as shown in <xref ref-type="table" rid="table-2">Table 2</xref>.</p>

<p>The &#x201C;FID&#x201D; measures the similarity between two images. It is evident that the FID score is relatively low, largely due to the VQGAN f16 artifacts being heavily penalized as highlighted in <xref ref-type="fig" rid="fig-8">Fig. 8</xref>. <xref ref-type="table" rid="table-3">Table 3</xref> displays the comparison of outputs of the suggested model with StackGAN&#x002B;&#x002B; and DALL-E. The observed output exhibits the CLIP&#x002B;VQGAN&#x2019;s capability to capture the exact images. It can be observed the closeness of the generated image according to prompt and learned images from the dataset provided. The ability of the CLIP &#x002B; VQGAN architecture to interpret and generate images based on speech input depends heavily on how well it processes and aligns the meaning of the input with visual representations. Abstract prompts (e.g., &#x201C;freedom,&#x201D; &#x201C;a dream of stars&#x201D;) and concrete prompts (e.g., &#x201C;a red car on a sunny road&#x201D;) present distinct challenges and opportunities for evaluation. As the model is trained on a limited dataset, which does not include diverse ranges of accents, this also degrades the model&#x2019;s performance. The data primarily consists of standard or regional accents, which may require the system to accurately recognize words spoken in different accents. This inhibition puts a limit on the performance of the model. However, accents can alter the pronunciation of words significantly. Different accents may emphasize or de-emphasize specific phonemes (sounds), leading to misinterpretation by the recognition algorithms. Different speech patterns, such as speed or clarity, can sometimes accompany accents. Another challenge is the systems can adapt to individual speakers over time. Still, this adaptation may only be effective for some accents, particularly if the system has yet to be trained in similar voices. <xref ref-type="table" rid="table-3">Table 3</xref> also showcases the borderline cases. The comparison of the model with standard techniques is shown in <xref ref-type="table" rid="table-4">Table 4</xref>. This gives the validation of the output generated by the suggested model. However, ImageNet is a comprehensive, labeled visual dataset that has been crucial in advancing computer vision research. Its vast collection of images across 1000 object categories has become a standard benchmark for training and evaluating models, particularly in tasks such as image classification and object detection.</p>
<fig id="fig-8">
<label>Figure 8</label>
<caption>
<title>Images generated by CLIP</title>
</caption>
<graphic mimetype="image" mime-subtype="tif" xlink:href="CMES_58456-fig-8.tif"/>
</fig>
<p>The suggested model produced decent results using less computational resources. Though the FID score is higher it could produce real-time images. FID is more due to poor image quality, mode collapse, inadequate training, feature mismatches, or a lack of diversity in either the generated images or the real dataset. Speech/audio related challenges also increase FID score. As the model is trained on a limited dataset, which does not include diverse ranges of accents, this also degrades the model&#x2019;s performance. The data primarily consists of standard or regional accents, which may require the system to accurately recognize words spoken in different accents. This inhibition puts a limit on the performance of the model. However, accents can alter the pronunciation of words significantly. Different accents may emphasize or de-emphasize specific phonemes (sounds), leading to misinterpretation by the recognition algorithms. Different speech patterns, such as speed or clarity, can sometimes accompany accents. Another challenge is the systems can adapt to individual speakers over time. Still, this adaptation may only be effective for some accents, particularly if the system has yet to be trained on similar voices.</p>
<p>The following are the limitations of this study:
<list list-type="bullet">
<list-item>
<p>CLIP often excels at identifying familiar things, but it suffers from more complex or systematic tasks. For highly fine-grained categorization, such as determining the differences between different automobile models, zero-shot CLIP performs poorly compared to task-specific models.</p></list-item>
<list-item>
<p>Additionally, CLIP still struggles to generalize to photos that were not included in its pretraining dataset.</p></list-item>
<list-item>
<p>CLIP&#x2019;s zero-shot classifiers can be delicate to phrasing and may require &#x0201C;prompt engineering&#x0201D; through trial and error to function successfully.</p></list-item>
<list-item>
<p>The speech prompts are translated to text using the Chrome Web-to-Speech API. This may not be fully accurate and can result in errors when translating words with various accents. The authors plan to follow up on this work by working on speech encoding prompts for image translation in the future.</p></list-item>
<list-item>
<p>The proposed model encounters challenges with ambiguous or unclear speech inputs, often resulting in less accurate or relevant image generation. For highly abstract concepts, the model may struggle to produce coherent visual outputs due to the complexity of translating abstract speech into visual elements.</p></list-item>
<list-item>
<p>One of the major challenges in this implementation was handling the computational demands associated with fine-tuning the CLIP and VQGAN models. Due to limited processing capacity, it was crucial to optimize the training process to ensure efficient use of resources. The employed techniques such as model pruning and mixed-precision training to reduce the computational load while maintaining the quality of the generated images.</p></list-item>
<list-item>
<p>Additionally, there are ethical considerations, such as the potential for misuse in generating misleading or inappropriate content. It is essential to implement guidelines and safeguards to ensure ethical use and address these challenges. Future research should focus on improving the model&#x2019;s robustness and developing ethical frameworks to guide its application.</p></list-item>
</list></p>
</sec>
<sec id="s7">
<label>7</label>
<title>Conclusion</title>
<p>This research introduced a framework for synthesizing images based on speech inputs, approached through three stages. Firstly, the model processes the voice input and extracts meaningful words using the speech module available in web browsers such as Chrome, Edge, and Firefox. Subsequently, the phrase is tokenized, and the model generates a text string with essential characteristics for the CLIP transformer to use as input. The image encoder utilizes zero-shot prediction to convert these tokens into a vector that the VQGAN can employ to produce image cropping. This technology holds promise for generating quasi-realistic graphic artistry and provides utility for the gaming industry in developing interactive components. The primary motivation for this research has been its potential applications in education, entertainment, and visual art fields.</p>
<p>Specifically, within the field of education, the model could be used to create visual aids for language learning, where students can generate images based on spoken descriptions, aiding in vocabulary acquisition and comprehension. In the field of art, artists could use the model to visualize their spoken ideas, fostering creative expression and exploration. In design, the model could assist designers in quickly prototyping visual concepts based on verbal descriptions, streamlining the design process. For instance, a case study in an art workshop demonstrated that participants were able to generate unique visual pieces by describing their concepts verbally, which the model then transformed into visual art. This will be a paradigm shift in these fields and would radically expand the boundaries of interface for artists, designers, and educators. The authors aim to enhance this study by focusing on audio-derived speech encoding and exploring alternative speech embedding techniques that could refine a bipartite model. The work can also be extended using stable diffusion models in different applications. Future research could focus on enhancing the model&#x2019;s ability to interpret and generate images from abstract or ambiguous prompts, possibly through advanced natural language processing techniques and more extensive training datasets.</p>
</sec>
</body>
<back>
<ack>
<p>Not applicable.</p>
</ack>
<sec>
<title>Funding Statement</title>
<p>This research was funded by the Centre for Advanced Modelling and Geospatial Information Systems (CAMGIS), Faculty of Engineering and IT, University of Technology Sydney. Moreover, supported by the Researchers Supporting Project, King Saud University, Riyadh, Saudi Arabia, under Ongoing Research Funding (ORF-2025-14).</p>
</sec>
<sec>
<title>Author Contributions</title>
<p>Conceptualization, Smita Mahajan, Shilpa Gite; methodology, Shaunak Inamdar, Deva Shriyansh, Akshat Ashish Shah, Shruti Agarwal, Shilpa Gite, Smita Mahajan; software, Shilpa Gite, Smita Mahajan; validation, Shaunak Inamdar, Deva Shriyansh, Akshat Ashish Shah, Shruti Agarwal, Shilpa Gite, Smita Mahajan, Biswajeet Pradhan, Abdullah Alamri; formal analysis, Shaunak Inamdar, Deva Shriyansh, Akshat Ashish Shah, Shruti Agarwal; investigation, Shaunak Inamdar, Deva Shriyansh, Akshat Ashish Shah, Shruti Agarwal, Shilpa Gite, Smita Mahajan, Biswajeet Pradhan, Abdullah Alamri; resources, Biswajeet Pradhan; data curation, Shaunak Inamdar, Deva Shriyansh, Akshat Ashish Shah, Shruti Agarwal; writing&#x2014;original draft preparation, Shaunak Inamdar, Deva Shriyansh, Akshat Ashish Shah, Shruti Agarwal; writing&#x2014;review and editing, Shilpa Gite, Smita Mahajan, Biswajeet Pradhan, Abdullah Alamri; visualization, Shilpa Gite, Smita Mahajan, Biswajeet Pradhan, Abdullah Alamri; supervision, Shilpa Gite, Smita Mahajan; project administration, Shilpa Gite, Smita Mahajan, Biswajeet Pradhan; funding acquisition, Biswajeet Pradhan, Abdullah Alamri. All authors reviewed the results and approved the final version of the manuscript.</p>
</sec>
<sec sec-type="data-availability">
<title>Availability of Data and Materials</title>
<p>ImageNet Dataset is having Free Access for Research. It is primarily intended for non-commercial academic research. Researchers need to create an account and request access to download the dataset. The link to the version of the code is available at the link provided here <ext-link ext-link-type="uri" xlink:href="https://colab.research.google.com/drive/1MJeP6z4opQC6sA9b20kjkh_HGXzFIyyV?usp=sharing">https://colab.research.google.com/drive/1MJeP6z4opQC6sA9b20kjkh_HGXzFIyyV?usp=sharing</ext-link> (accessed on 25 January 2025).</p>
</sec>
<sec>
<title>Ethics Approval</title>
<p>Not applicable.</p>
</sec>
<sec sec-type="COI-statement">
<title>Conflicts of Interest</title>
<p>The authors declare no conflicts of interest to report regarding the present study.</p>
</sec>
<ref-list content-type="authoryear">
<title>References</title>
<ref id="ref-1"><label>[1]</label><mixed-citation publication-type="conf-proc"><person-group person-group-type="author"><string-name><surname>Siarohin</surname> <given-names>A</given-names></string-name>, <string-name><surname>Sangineto</surname> <given-names>E</given-names></string-name>, <string-name><surname>Lathuili&#x00E8;re</surname> <given-names>S</given-names></string-name>, <string-name><surname>Sebe</surname> <given-names>N</given-names></string-name></person-group>. <article-title>Deformable GANs for pose-based human image generation</article-title>. In: <conf-name>IEEE/CVF Conference on Computer Vision and Pattern Recognition</conf-name>; <year>2018 Jun 18&#x2013;23</year>; <publisher-loc>Salt Lake City, UT, USA</publisher-loc>: <publisher-name>IEEE</publisher-name>; 2018. p. <fpage>3408</fpage>&#x2013;<lpage>16</lpage>. doi:<pub-id pub-id-type="doi">10.1109/CVPR.2018.00359</pub-id>.</mixed-citation></ref>
<ref id="ref-2"><label>[2]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><surname>Baraheem</surname> <given-names>SS</given-names></string-name>, <string-name><surname>Le</surname> <given-names>TN</given-names></string-name>, <string-name><surname>Nguyen</surname> <given-names>TV</given-names></string-name></person-group>. <article-title>Image synthesis: a review of methods, datasets, evaluation metrics, and future outlook</article-title>. <source>Artif Intell Rev</source>. <year>2023</year>;<volume>56</volume>(<issue>10</issue>):<fpage>10813</fpage>&#x2013;<lpage>65</lpage>. doi:<pub-id pub-id-type="doi">10.1007/s10462-023-10434-2</pub-id>.</mixed-citation></ref>
<ref id="ref-3"><label>[3]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><surname>Liz-L&#x00F3;pez</surname> <given-names>H</given-names></string-name>, <string-name><surname>Keita</surname> <given-names>M</given-names></string-name>, <string-name><surname>Taleb-Ahmed</surname> <given-names>A</given-names></string-name>, <string-name><surname>Hadid</surname> <given-names>A</given-names></string-name>, <string-name><surname>Huertas-Tato</surname> <given-names>J</given-names></string-name>, <string-name><surname>Camacho</surname> <given-names>D</given-names></string-name></person-group>. <article-title>Generation and detection of manipulated multimodal audiovisual content: advances, trends and open challenges</article-title>. <source>Inf Fusion</source>. <year>2024</year>;<volume>103</volume>(<issue>2</issue>):<fpage>102103</fpage>. doi:<pub-id pub-id-type="doi">10.1016/j.inffus.2023.102103</pub-id>.</mixed-citation></ref>
<ref id="ref-4"><label>[4]</label><mixed-citation publication-type="conf-proc"><person-group person-group-type="author"><string-name><surname>Arjovsky</surname> <given-names>M</given-names></string-name>, <string-name><surname>Chintala</surname> <given-names>S</given-names></string-name>, <string-name><surname>Bottou</surname> <given-names>L</given-names></string-name>, <string-name><surname>Arjovsky</surname> <given-names>M</given-names></string-name>, <string-name><surname>Chintala</surname> <given-names>S</given-names></string-name>, <string-name><surname>Bottou</surname> <given-names>L</given-names></string-name></person-group>. <article-title>Wasserstein generative adversarial networks</article-title>. In: <conf-name>Proceedings of the 34th International Conference on Machine Learning</conf-name>; <year>2017 Aug 6&#x2013;11</year>; <publisher-loc>Sydney, NSW, Australia</publisher-loc>: <publisher-name>ACM</publisher-name>; 2017. p. <fpage>214</fpage>&#x2013;<lpage>23</lpage>. doi:<pub-id pub-id-type="doi">10.5555/3305381.3305404</pub-id>.</mixed-citation></ref>
<ref id="ref-5"><label>[5]</label><mixed-citation publication-type="other"><person-group person-group-type="author"><string-name><surname>Andonian</surname> <given-names>A</given-names></string-name>, <string-name><surname>Osmany</surname> <given-names>S</given-names></string-name>, <string-name><surname>Cui</surname> <given-names>A</given-names></string-name>, <string-name><surname>Park</surname> <given-names>Y</given-names></string-name>, <string-name><surname>Jahanian</surname> <given-names>A</given-names></string-name>, <string-name><surname>Torralba</surname> <given-names>A</given-names></string-name>, <etal>et al.</etal></person-group> <article-title>Paint by word</article-title>. <comment>arXiv:2103.10951. 2021</comment>.</mixed-citation></ref>
<ref id="ref-6"><label>[6]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><surname>Zhang</surname> <given-names>H</given-names></string-name>, <string-name><surname>Xu</surname> <given-names>T</given-names></string-name>, <string-name><surname>Li</surname> <given-names>H</given-names></string-name>, <string-name><surname>Zhang</surname> <given-names>S</given-names></string-name>, <string-name><surname>Wang</surname> <given-names>X</given-names></string-name>, <string-name><surname>Huang</surname> <given-names>X</given-names></string-name>, <etal>et al</etal></person-group>. <article-title>StackGAN: realistic image synthesis with stacked generative adversarial networks</article-title>. <source>IEEE Trans Pattern Anal Mach Intell</source>. <year>2019</year>;<volume>41</volume>(<issue>8</issue>):<fpage>1947</fpage>&#x2013;<lpage>62</lpage>. doi:<pub-id pub-id-type="doi">10.1109/TPAMI.2018.2856256</pub-id>; <pub-id pub-id-type="pmid">30010548</pub-id></mixed-citation></ref>
<ref id="ref-7"><label>[7]</label><mixed-citation publication-type="other"><person-group person-group-type="author"><string-name><surname>Brock</surname> <given-names>A</given-names></string-name>, <string-name><surname>Donahue</surname> <given-names>J</given-names></string-name>, <string-name><surname>Simonyan</surname> <given-names>K</given-names></string-name></person-group>. <article-title>Large scale GAN training for high fidelity natural image synthesis</article-title>. <comment>arXiv:1809.11096. 2018</comment>.</mixed-citation></ref>
<ref id="ref-8"><label>[8]</label><mixed-citation publication-type="book"><person-group person-group-type="author"><string-name><surname>Lin</surname> <given-names>ZY</given-names></string-name>, <string-name><surname>Geng</surname> <given-names>S</given-names></string-name>, <string-name><surname>Zhang</surname> <given-names>R</given-names></string-name>, <string-name><surname>Gao</surname> <given-names>P</given-names></string-name>, <string-name><surname>De Melo</surname> <given-names>G</given-names></string-name>, <string-name><surname>Wang</surname> <given-names>X</given-names></string-name>, <etal>et al</etal></person-group>. <chapter-title>Frozen CLIP models are efficient video learners</chapter-title>. In: <source>Computer Vision&#x2014;ECCV 2022</source>: <conf-name>17th European Conference</conf-name>; <year>2022 Oct 23&#x2013;27</year>; <publisher-loc>Tel Aviv, Israel</publisher-loc>. <year>2022</year>. p. <fpage>388</fpage>&#x2013;<lpage>404</lpage>. doi:<pub-id pub-id-type="doi">10.1007/978-3-031-19833-5_23</pub-id>.</mixed-citation></ref>
<ref id="ref-9"><label>[9]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><surname>Li</surname> <given-names>J</given-names></string-name>, <string-name><surname>Tang</surname> <given-names>T</given-names></string-name>, <string-name><surname>Zhao</surname> <given-names>WX</given-names></string-name>, <string-name><surname>Nie</surname> <given-names>JY</given-names></string-name>, <string-name><surname>Wen</surname> <given-names>JR</given-names></string-name></person-group>. <article-title>Pre-trained language models for text generation: a survey</article-title>. <source>ACM Comput Surv</source>. <year>2024</year>;<volume>56</volume>(<issue>9</issue>):<fpage>1</fpage>&#x2013;<lpage>39</lpage>. doi:<pub-id pub-id-type="doi">10.1145/3649449</pub-id>.</mixed-citation></ref>
<ref id="ref-10"><label>[10]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><surname>Brown</surname> <given-names>T</given-names></string-name>, <string-name><surname>Mann</surname> <given-names>B</given-names></string-name>, <string-name><surname>Ryder</surname> <given-names>N</given-names></string-name>, <string-name><surname>Subbiah</surname> <given-names>M</given-names></string-name>, <string-name><surname>Kaplan</surname> <given-names>JD</given-names></string-name>, <string-name><surname>Dhariwal</surname> <given-names>P</given-names></string-name>, <etal>et al.</etal></person-group> <article-title>Language models are few-shot learners</article-title>. <source>Adv Neural Inform Process Syst</source>. <year>2020</year>;<volume>33</volume>:<fpage>1877</fpage>&#x2013;<lpage>901</lpage>.</mixed-citation></ref>
<ref id="ref-11"><label>[11]</label><mixed-citation publication-type="conf-proc"><person-group person-group-type="author"><string-name><surname>Chen</surname> <given-names>M</given-names></string-name>, <string-name><surname>Radford</surname> <given-names>A</given-names></string-name>, <string-name><surname>Child</surname> <given-names>R</given-names></string-name>, <string-name><surname>Wu</surname> <given-names>J</given-names></string-name>, <string-name><surname>Jun</surname> <given-names>H</given-names></string-name>, <string-name><surname>Luan</surname> <given-names>D</given-names></string-name>, <etal>et al.</etal></person-group> <article-title>Generative pretraining from pixels</article-title>. In: <conf-name>Proceedings of the 37th International Conference on Machine Learning</conf-name>; <year>2020</year>; <publisher-loc>Vienna, Austria</publisher-loc>. p. <fpage>1691</fpage>&#x2013;<lpage>703</lpage>.</mixed-citation></ref>
<ref id="ref-12"><label>[12]</label><mixed-citation publication-type="conf-proc"><person-group person-group-type="author"><string-name><surname>Dolhansky</surname> <given-names>B</given-names></string-name>, <string-name><surname>Ferrer</surname> <given-names>CC</given-names></string-name></person-group>. <article-title>Eye in-painting with exemplar generative adversarial networks</article-title>. In: <conf-name>2018 IEEE/CVF Conference on Computer Vision and Pattern Recognition</conf-name>; <year>2018 Jun 18&#x2013;23</year>; <publisher-loc>Salt Lake City, UT, USA</publisher-loc>: <publisher-name>IEEE</publisher-name>; 2018. p. <fpage>7902</fpage>&#x2013;<lpage>11</lpage>. doi:<pub-id pub-id-type="doi">10.1109/CVPR.2018.00824</pub-id>.</mixed-citation></ref>
<ref id="ref-13"><label>[13]</label><mixed-citation publication-type="conf-proc"><person-group person-group-type="author"><string-name><surname>Esser</surname> <given-names>P</given-names></string-name>, <string-name><surname>Rombach</surname> <given-names>R</given-names></string-name>, <string-name><surname>Ommer</surname> <given-names>B</given-names></string-name></person-group>. <article-title>Taming transformers for high-resolution image synthesis</article-title>. In: <conf-name>IEEE/CVF Conference on Computer Vision and Pattern Recognition (CVPR)</conf-name>; <year>2021 Jun 20&#x2013;25</year>; <publisher-loc>Nashville, TN, USA</publisher-loc>: <publisher-name>IEEE</publisher-name>; 2021. p. <fpage>12868</fpage>&#x2013;<lpage>78</lpage>. doi:<pub-id pub-id-type="doi">10.1109/cvpr46437.2021.01268</pub-id>.</mixed-citation></ref>
<ref id="ref-14"><label>[14]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><surname>Frolov</surname> <given-names>S</given-names></string-name>, <string-name><surname>Hinz</surname> <given-names>T</given-names></string-name>, <string-name><surname>Raue</surname> <given-names>F</given-names></string-name>, <string-name><surname>Hees</surname> <given-names>J</given-names></string-name>, <string-name><surname>Dengel</surname> <given-names>A</given-names></string-name></person-group>. <article-title>Adversarial text-to-image synthesis: a review</article-title>. <source>Neural Netw</source>. <year>2021</year>;<volume>144</volume>(<issue>4</issue>):<fpage>187</fpage>&#x2013;<lpage>209</lpage>. doi:<pub-id pub-id-type="doi">10.1016/j.neunet.2021.07.019</pub-id>; <pub-id pub-id-type="pmid">34500257</pub-id></mixed-citation></ref>
<ref id="ref-15"><label>[15]</label><mixed-citation publication-type="conf-proc"><person-group person-group-type="author"><string-name><surname>Ge</surname> <given-names>S</given-names></string-name>, <string-name><surname>Hayes</surname> <given-names>T</given-names></string-name>, <string-name><surname>Yang</surname> <given-names>H</given-names></string-name>, <string-name><surname>Yin</surname> <given-names>X</given-names></string-name>, <string-name><surname>Pang</surname> <given-names>G</given-names></string-name>, <string-name><surname>Jacobs</surname> <given-names>D</given-names></string-name>, <etal>et al</etal></person-group>. <article-title>Long video generation with time-agnostic VQGAN and time-sensitive transformer</article-title>. In: <conf-name>Computer Vision&#x2013;ECCV 2022: 17th European Conference</conf-name>; <year>2022 Oct 23&#x2013;27</year>; <publisher-loc>Tel Aviv, Israel</publisher-loc>. <year>2022</year>. p. <fpage>102</fpage>&#x2013;<lpage>18</lpage>.</mixed-citation></ref>
<ref id="ref-16"><label>[16]</label><mixed-citation publication-type="other"><person-group person-group-type="author"><string-name><surname>Goodfellow</surname> <given-names>IJ</given-names></string-name>, <string-name><surname>Pouget-Abadie</surname> <given-names>J</given-names></string-name>, <string-name><surname>Mirza</surname> <given-names>M</given-names></string-name>, <string-name><surname>Xu</surname> <given-names>B</given-names></string-name>, <string-name><surname>Warde-Farley</surname> <given-names>D</given-names></string-name>, <string-name><surname>Ozair</surname> <given-names>S</given-names></string-name>, <etal>et al.</etal></person-group> <article-title>Generative adversarial nets</article-title>. <comment>arXiv:1406.2661. 2014</comment>.</mixed-citation></ref>
<ref id="ref-17"><label>[17]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><surname>Ling</surname> <given-names>H</given-names></string-name>, <string-name><surname>Kreis</surname> <given-names>K</given-names></string-name>, <string-name><surname>Li</surname> <given-names>D</given-names></string-name>, <string-name><surname>Kim</surname> <given-names>SW</given-names></string-name>, <string-name><surname>Torralba</surname> <given-names>A</given-names></string-name>, <string-name><surname>Fidler</surname> <given-names>S</given-names></string-name></person-group>. <article-title>EditGAN: high-precision semantic image editing</article-title>. <source>Adv Neural Inform Process Syst</source>. <year>2021</year>;<volume>34</volume>:<fpage>16331</fpage>&#x2013;<lpage>45</lpage>.</mixed-citation></ref>
<ref id="ref-18"><label>[18]</label><mixed-citation publication-type="other"><person-group person-group-type="author"><string-name><surname>Donahue</surname> <given-names>J</given-names></string-name>, <string-name><surname>Dieleman</surname> <given-names>S</given-names></string-name>, <string-name><surname>Bi&#x0144;kowski</surname> <given-names>M</given-names></string-name>, <string-name><surname>Elsen</surname> <given-names>E</given-names></string-name>, <string-name><surname>Simonyan</surname> <given-names>K</given-names></string-name></person-group>. <article-title>End-to-end adversarial text-to-speech</article-title>. <comment>arXiv:2006.03575. 2020</comment>.</mixed-citation></ref>
<ref id="ref-19"><label>[19]</label><mixed-citation publication-type="conf-proc"><person-group person-group-type="author"><string-name><surname>Deng</surname> <given-names>J</given-names></string-name>, <string-name><surname>Dong</surname> <given-names>W</given-names></string-name>, <string-name><surname>Socher</surname> <given-names>R</given-names></string-name>, <string-name><surname>Li</surname> <given-names>LJ</given-names></string-name>, <string-name><surname>Kai</surname> <given-names>L</given-names></string-name>, <string-name><surname>Li</surname> <given-names>FF</given-names></string-name></person-group>. <article-title>ImageNet: a large-scale hierarchical image database</article-title>. In: <conf-name>2009 IEEE Conference on Computer Vision and Pattern Recognition</conf-name>; <year>2009 Jun 20&#x2013;25</year>; <publisher-loc>Miami, FL, USA</publisher-loc>: <publisher-name>IEEE</publisher-name>; 2009. p. <fpage>248</fpage>&#x2013;<lpage>55</lpage>. doi:<pub-id pub-id-type="doi">10.1109/CVPR.2009.5206848</pub-id>.</mixed-citation></ref>
<ref id="ref-20"><label>[20]</label><mixed-citation publication-type="conf-proc"><person-group person-group-type="author"><string-name><surname>Park</surname> <given-names>K</given-names></string-name>, <string-name><surname>Mott</surname> <given-names>BW</given-names></string-name>, <string-name><surname>Min</surname> <given-names>W</given-names></string-name>, <string-name><surname>Boyer</surname> <given-names>KE</given-names></string-name>, <string-name><surname>Wiebe</surname> <given-names>EN</given-names></string-name>, <string-name><surname>Lester</surname> <given-names>JC</given-names></string-name></person-group>. <article-title>Generating educational game levels with multistep deep convolutional generative adversarial networks</article-title>. In: <conf-name>IEEE Conference on Games (CoG)</conf-name>; <year>2019 Aug 20&#x2013;23</year>; <publisher-loc>London, UK</publisher-loc>: <publisher-name>IEEE</publisher-name>; 2019. p. <fpage>1</fpage>&#x2013;<lpage>8</lpage>. doi:<pub-id pub-id-type="doi">10.1109/cig.2019.8848085</pub-id>.</mixed-citation></ref>
<ref id="ref-21"><label>[21]</label><mixed-citation publication-type="conf-proc"><person-group person-group-type="author"><string-name><surname>Crowson</surname> <given-names>K</given-names></string-name>, <string-name><surname>Biderman</surname> <given-names>S</given-names></string-name>, <string-name><surname>Kornis</surname> <given-names>D</given-names></string-name>, <string-name><surname>Stander</surname> <given-names>D</given-names></string-name>, <string-name><surname>Hallahan</surname> <given-names>E</given-names></string-name>, <string-name><surname>Castricato</surname> <given-names>L</given-names></string-name>, <etal>et al.</etal></person-group> <article-title>VQGAN-CLIP: open domain image generation and editing with natural language guidance</article-title>. In: <conf-name>Computer Vision&#x2013;ECCV 2022</conf-name>; <year>2022</year>; <publisher-loc>Cham, Switzerland</publisher-loc>: <publisher-name>Springer Nature Switzerland</publisher-name>. p. <fpage>88</fpage>&#x2013;<lpage>105</lpage>. doi:<pub-id pub-id-type="doi">10.1007/978-3-031-19836-6_6</pub-id>.</mixed-citation></ref>
<ref id="ref-22"><label>[22]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><surname>Orynbay</surname> <given-names>L</given-names></string-name>, <string-name><surname>Razakhova</surname> <given-names>B</given-names></string-name>, <string-name><surname>Peer</surname> <given-names>P</given-names></string-name>, <string-name><surname>Meden</surname> <given-names>B</given-names></string-name>, <string-name><surname>Emer&#x0161;i&#x010D;</surname> <given-names>&#x017D;</given-names></string-name></person-group>. <article-title>Recent advances in synthesis and interaction of speech, text, and vision</article-title>. <source>Electronics</source>. <year>2024</year>;<volume>13</volume>(<issue>9</issue>):<fpage>1726</fpage>. doi:<pub-id pub-id-type="doi">10.3390/electronics13091726</pub-id>.</mixed-citation></ref>
<ref id="ref-23"><label>[23]</label><mixed-citation publication-type="conf-proc"><person-group person-group-type="author"><string-name><surname>Karras</surname> <given-names>T</given-names></string-name>, <string-name><surname>Laine</surname> <given-names>S</given-names></string-name>, <string-name><surname>Aittala</surname> <given-names>M</given-names></string-name>, <string-name><surname>Hellsten</surname> <given-names>J</given-names></string-name>, <string-name><surname>Lehtinen</surname> <given-names>J</given-names></string-name>, <string-name><surname>Aila</surname> <given-names>T</given-names></string-name></person-group>. <article-title>Analyzing and improving the image quality of StyleGAN</article-title>. In: <conf-name>2020 IEEE/CVF Conference on Computer Vision and Pattern Recognition (CVPR)</conf-name>; <year>2020 Jun 13&#x2013;19</year>; <publisher-loc>Seattle, WA, USA</publisher-loc>: <publisher-name>IEEE</publisher-name>; 2020. p. <fpage>8107</fpage>&#x2013;<lpage>16</lpage>. doi:<pub-id pub-id-type="doi">10.1109/cvpr42600.2020.00813</pub-id>.</mixed-citation></ref>
<ref id="ref-24"><label>[24]</label><mixed-citation publication-type="other"><person-group person-group-type="author"><string-name><surname>Karras</surname> <given-names>T</given-names></string-name>, <string-name><surname>Aila</surname> <given-names>T</given-names></string-name>, <string-name><surname>Laine</surname> <given-names>S</given-names></string-name>, <string-name><surname>Lehtinen</surname> <given-names>J</given-names></string-name></person-group>. <article-title>Progressive growing of GANs for improved quality, stability, and variation</article-title>. <comment>arXiv:1710.10196. 2017</comment>.</mixed-citation></ref>
<ref id="ref-25"><label>[25]</label><mixed-citation publication-type="conf-proc"><person-group person-group-type="author"><string-name><surname>Karras</surname> <given-names>T</given-names></string-name>, <string-name><surname>Laine</surname> <given-names>S</given-names></string-name>, <string-name><surname>Aila</surname> <given-names>T</given-names></string-name></person-group>. <article-title>A style-based generator architecture for generative adversarial networks</article-title>. In: <conf-name>IEEE/CVF Conference on Computer Vision and Pattern Recognition (CVPR)</conf-name>; <year>2019 Jun 15&#x2013;20</year>; <publisher-loc>Long Beach, CA, USA</publisher-loc>: <publisher-name>IEEE</publisher-name>; 2019. p. <fpage>4401</fpage>&#x2013;<lpage>10</lpage>. doi:<pub-id pub-id-type="doi">10.1109/cvpr.2019.00453</pub-id>.</mixed-citation></ref>
<ref id="ref-26"><label>[26]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><surname>Khan</surname> <given-names>S</given-names></string-name>, <string-name><surname>Naseer</surname> <given-names>M</given-names></string-name>, <string-name><surname>Hayat</surname> <given-names>M</given-names></string-name>, <string-name><surname>Zamir</surname> <given-names>SW</given-names></string-name>, <string-name><surname>Khan</surname> <given-names>FS</given-names></string-name>, <string-name><surname>Shah</surname> <given-names>M</given-names></string-name></person-group>. <article-title>Transformers in vision: a survey</article-title>. <source>ACM Comput Surv</source>. <year>2022</year>;<volume>54</volume>(<issue>10s</issue>):<fpage>1</fpage>&#x2013;<lpage>41</lpage>. doi:<pub-id pub-id-type="doi">10.1145/3505244</pub-id>.</mixed-citation></ref>
<ref id="ref-27"><label>[27]</label><mixed-citation publication-type="other"><person-group person-group-type="author"><string-name><surname>Li</surname> <given-names>XL</given-names></string-name>, <string-name><surname>Liang</surname> <given-names>P</given-names></string-name></person-group>. <article-title>Prefix-tuning: optimizing continuous prompts for generation</article-title>. <comment>arXiv:2101.00190. 2021</comment>.</mixed-citation></ref>
<ref id="ref-28"><label>[28]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><surname>Li</surname> <given-names>X</given-names></string-name>, <string-name><surname>Zhang</surname> <given-names>J</given-names></string-name>, <string-name><surname>Liu</surname> <given-names>Y</given-names></string-name></person-group>. <article-title>Speech driven facial animation generation based on GAN</article-title>. <source>Displays</source>. <year>2022</year>;<volume>74</volume>:<fpage>102260</fpage>. doi:<pub-id pub-id-type="doi">10.1016/j.displa.2022.102260</pub-id>.</mixed-citation></ref>
<ref id="ref-29"><label>[29]</label><mixed-citation publication-type="conf-proc"><person-group person-group-type="author"><string-name><surname>Meyer</surname> <given-names>S</given-names></string-name>, <string-name><surname>Tilli</surname> <given-names>P</given-names></string-name>, <string-name><surname>Denisov</surname> <given-names>P</given-names></string-name>, <string-name><surname>Lux</surname> <given-names>F</given-names></string-name>, <string-name><surname>Koch</surname> <given-names>J</given-names></string-name>, <string-name><surname>Vu</surname> <given-names>NT</given-names></string-name></person-group>. <article-title>Anonymizing speech with generative adversarial networks to preserve speaker privacy</article-title>. In: <conf-name>2022 IEEE Spoken Language Technology Workshop (SLT)</conf-name>; <year>2023 Jan 9&#x2013;12</year>; <publisher-loc>Doha, Qatar</publisher-loc>: <publisher-name>IEEE</publisher-name>; 2023. p. <fpage>912</fpage>&#x2013;<lpage>9</lpage>. doi:<pub-id pub-id-type="doi">10.1109/SLT54892.2023.10022601</pub-id>.</mixed-citation></ref>
<ref id="ref-30"><label>[30]</label><mixed-citation publication-type="other"><person-group person-group-type="author"><string-name><surname>Liu</surname> <given-names>MY</given-names></string-name>, <string-name><surname>Tuzel</surname> <given-names>O</given-names></string-name></person-group>. <article-title>Coupled generative adversarial networks</article-title>. <comment>arXiv:1606.07536. 2016</comment>.</mixed-citation></ref>
<ref id="ref-31"><label>[31]</label><mixed-citation publication-type="other"><person-group person-group-type="author"><string-name><surname>Mansimov</surname> <given-names>E</given-names></string-name>, <string-name><surname>Parisotto</surname> <given-names>E</given-names></string-name>, <string-name><surname>Ba</surname> <given-names>JL</given-names></string-name>, <string-name><surname>Salakhutdinov</surname> <given-names>R</given-names></string-name></person-group>. <article-title>Generating images from captions with attention</article-title>. <comment>arXiv:1511.02793. 2015</comment>.</mixed-citation></ref>
<ref id="ref-32"><label>[32]</label><mixed-citation publication-type="other"><person-group person-group-type="author"><string-name><surname>Mirza</surname> <given-names>M</given-names></string-name>, <string-name><surname>Osindero</surname> <given-names>S</given-names></string-name></person-group>. <article-title>Conditional generative adversarial nets</article-title>. <comment>arXiv:1411.1784. 2014</comment>.</mixed-citation></ref>
<ref id="ref-33"><label>[33]</label><mixed-citation publication-type="other"><person-group person-group-type="author"><string-name><surname>Mokady</surname> <given-names>R</given-names></string-name>, <string-name><surname>Hertz</surname> <given-names>A</given-names></string-name>, <string-name><surname>Bermano</surname> <given-names>AH</given-names></string-name></person-group>. <article-title>ClipCap: clip prefix for image captioning</article-title>. <comment>arXiv:2111.09734. 2021</comment>.</mixed-citation></ref>
<ref id="ref-34"><label>[34]</label><mixed-citation publication-type="conf-proc"><person-group person-group-type="author"><string-name><surname>Radford</surname> <given-names>A</given-names></string-name>, <string-name><surname>Kim</surname> <given-names>JW</given-names></string-name>, <string-name><surname>Hallacy</surname> <given-names>C</given-names></string-name>, <string-name><surname>Ramesh</surname> <given-names>A</given-names></string-name>, <string-name><surname>Goh</surname> <given-names>G</given-names></string-name>, <string-name><surname>Agarwal</surname> <given-names>S</given-names></string-name>, <etal>et al.</etal></person-group> <article-title>Learning transferable visual models from natural language supervision</article-title>. In: <conf-name>Proceedings of the International Conference on Machine Learning (ICML)</conf-name>; <year>2021</year>. Vol. <volume>139</volume>, p. <fpage>8748</fpage>&#x2013;<lpage>63</lpage>.</mixed-citation></ref>
<ref id="ref-35"><label>[35]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><surname>Saharia</surname> <given-names>C</given-names></string-name>, <string-name><surname>Chan</surname> <given-names>W</given-names></string-name>, <string-name><surname>Saxena</surname> <given-names>S</given-names></string-name>, <string-name><surname>Li</surname> <given-names>L</given-names></string-name>, <string-name><surname>Whang</surname> <given-names>J</given-names></string-name>, <string-name><surname>Denton</surname> <given-names>EL</given-names></string-name>, <etal>et al</etal></person-group>. <article-title>Photorealistic text-to-image diffusion models with deep language understanding</article-title>. <source>Adv Neural Inf Process Syst</source>. <year>2022</year>;<volume>35</volume>:<fpage>36479</fpage>&#x2013;<lpage>94</lpage>.</mixed-citation></ref>
<ref id="ref-36"><label>[36]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><surname>Li</surname> <given-names>H</given-names></string-name>, <string-name><surname>Xu</surname> <given-names>F</given-names></string-name>, <string-name><surname>Lin</surname> <given-names>Z</given-names></string-name></person-group>. <article-title>ET-DM: text to image via diffusion model with efficient Transformer</article-title>. <source>Displays</source>. <year>2023</year>;<volume>80</volume>(<issue>1</issue>):<fpage>102568</fpage>. doi:<pub-id pub-id-type="doi">10.1016/j.displa.2023.102568</pub-id>.</mixed-citation></ref>
<ref id="ref-37"><label>[37]</label><mixed-citation publication-type="other"><person-group person-group-type="author"><string-name><surname>Chen</surname> <given-names>J</given-names></string-name>, <string-name><surname>Yu</surname> <given-names>J</given-names></string-name>, <string-name><surname>Ge</surname> <given-names>C</given-names></string-name>, <string-name><surname>Yao</surname> <given-names>L</given-names></string-name>, <string-name><surname>Xie</surname> <given-names>E</given-names></string-name>, <string-name><surname>Wu</surname> <given-names>Y</given-names></string-name>, <etal>et al</etal></person-group>. <article-title>PixArt-&#x03B1;: fast training of diffusion transformer for photorealistic text-to-image synthesis</article-title>. <comment>arXiv: 2310.00426. 2023</comment>.</mixed-citation></ref>
<ref id="ref-38"><label>[38]</label><mixed-citation publication-type="conf-proc"><person-group person-group-type="author"><string-name><surname>Seneviratne</surname> <given-names>S</given-names></string-name>, <string-name><surname>Senanayake</surname> <given-names>D</given-names></string-name>, <string-name><surname>Rasnayaka</surname> <given-names>S</given-names></string-name>, <string-name><surname>Vidanaarachchi</surname> <given-names>R</given-names></string-name>, <string-name><surname>Thompson</surname> <given-names>J</given-names></string-name></person-group>. <article-title>DALLE-URBAN: capturing the urban design expertise of large text to image transformers</article-title>. In: <conf-name>2022 International Conference on Digital Image Computing: Techniques and Applications (DICTA)</conf-name>; <year>2022 Nov 30&#x2013;Dec 2</year>; <publisher-loc>Sydney, Australia</publisher-loc>: <publisher-name>IEEE</publisher-name>; 2022. p. <fpage>1</fpage>&#x2013;<lpage>9</lpage>. doi:<pub-id pub-id-type="doi">10.1109/DICTA56598.2022.10034603</pub-id>.</mixed-citation></ref>
<ref id="ref-39"><label>[39]</label><mixed-citation publication-type="other"><person-group person-group-type="author"><string-name><surname>Radford</surname> <given-names>A</given-names></string-name>, <string-name><surname>Metz</surname> <given-names>L</given-names></string-name>, <string-name><surname>Chintala</surname> <given-names>S</given-names></string-name>, <string-name><surname>Dinakaran</surname> <given-names>R</given-names></string-name>, <string-name><surname>Easom</surname> <given-names>P</given-names></string-name>, <string-name><surname>Zhang</surname> <given-names>L</given-names></string-name>, <etal>et al</etal></person-group>. <article-title>Unsupervised representation learning with deep convolutional generative adversarial networks</article-title>. <comment>arXiv:1511.06434. 2015</comment>.</mixed-citation></ref>
<ref id="ref-40"><label>[40]</label><mixed-citation publication-type="conf-proc"><person-group person-group-type="author"><string-name><surname>Patashnik</surname> <given-names>O</given-names></string-name>, <string-name><surname>Wu</surname> <given-names>Z</given-names></string-name>, <string-name><surname>Shechtman</surname> <given-names>E</given-names></string-name>, <string-name><surname>Cohen-Or</surname> <given-names>D</given-names></string-name>, <string-name><surname>Lischinski</surname> <given-names>D</given-names></string-name></person-group>. <article-title>StyleCLIP: text-driven manipulation of StyleGAN imagery</article-title>. In: <conf-name>2021 IEEE/CVF International Conference on Computer Vision (ICCV)</conf-name>; <year>2021 Oct 10&#x2013;17</year>; <publisher-loc>Montreal, QC, Canada</publisher-loc>: <publisher-name>IEEE</publisher-name>; 2021. p. <fpage>2085</fpage>&#x2013;<lpage>94</lpage>. doi:<pub-id pub-id-type="doi">10.1109/ICCV48922.2021.00209</pub-id>.</mixed-citation></ref>
<ref id="ref-41"><label>[41]</label><mixed-citation publication-type="conf-proc"><person-group person-group-type="author"><string-name><surname>Ramesh</surname> <given-names>A</given-names></string-name>, <string-name><surname>Pavlov</surname> <given-names>M</given-names></string-name>, <string-name><surname>Goh</surname> <given-names>G</given-names></string-name>, <string-name><surname>Gray</surname> <given-names>S</given-names></string-name>, <string-name><surname>Voss</surname> <given-names>C</given-names></string-name>, <string-name><surname>Radford</surname> <given-names>A</given-names></string-name>, <etal>et al.</etal></person-group> <article-title>Zero-shot text-to-image generation</article-title>. In: <conf-name>Proceedings of the International Conference on Machine Learning (ICML)</conf-name>; <year>2021</year>. Vol. <volume>139</volume>, p. <fpage>8821</fpage>&#x2013;<lpage>31</lpage>.</mixed-citation></ref>
<ref id="ref-42"><label>[42]</label><mixed-citation publication-type="conf-proc"><person-group person-group-type="author"><string-name><surname>Reed</surname> <given-names>S</given-names></string-name>, <string-name><surname>Akata</surname> <given-names>Z</given-names></string-name>, <string-name><surname>Yan</surname> <given-names>X</given-names></string-name>, <string-name><surname>Logeswaran</surname> <given-names>L</given-names></string-name>, <string-name><surname>Schiele</surname> <given-names>B</given-names></string-name>, <string-name><surname>Lee</surname> <given-names>H</given-names></string-name></person-group>. <article-title>Generative adversarial text-to-image synthesis</article-title>. In: <conf-name>Proceedings of the International Conference on Machine Learning (ICML)</conf-name>; <year>2016</year>. Vol. <volume>49</volume>, p. <fpage>1060</fpage>&#x2013;<lpage>9</lpage>.</mixed-citation></ref>
<ref id="ref-43"><label>[43]</label><mixed-citation publication-type="other"><person-group person-group-type="author"><string-name><surname>Song</surname> <given-names>H</given-names></string-name>, <string-name><surname>Dong</surname> <given-names>L</given-names></string-name>, <string-name><surname>Zhang</surname> <given-names>WN</given-names></string-name>, <string-name><surname>Liu</surname> <given-names>T</given-names></string-name>, <string-name><surname>Wei</surname> <given-names>F</given-names></string-name></person-group>. <article-title>CLIP models are few-shot learners: empirical studies on VQA and visual entailment</article-title>. <comment>arXiv:2203.07190. 2022</comment>.</mixed-citation></ref>
<ref id="ref-44"><label>[44]</label><mixed-citation publication-type="book"><person-group person-group-type="author"><string-name><surname>Vaswani</surname> <given-names>A</given-names></string-name>, <string-name><surname>Shazeer</surname> <given-names>N</given-names></string-name>, <string-name><surname>Parmar</surname> <given-names>N</given-names></string-name>, <string-name><surname>Uszkoreit</surname> <given-names>J</given-names></string-name>, <string-name><surname>Jones</surname> <given-names>L</given-names></string-name>, <string-name><surname>Gomez</surname> <given-names>AN</given-names></string-name>, <etal>et al</etal></person-group>. <chapter-title>Attention is all you need</chapter-title>. In: <source>Advances in Neural Information Processing Systems (NeurIPS 2017)</source>; <year>2017</year>; <publisher-name>Curran Associates, Inc</publisher-name>. Vol. <volume>30</volume>.</mixed-citation></ref>
<ref id="ref-45"><label>[45]</label><mixed-citation publication-type="conf-proc"><person-group person-group-type="author"><string-name><surname>Tulyakov</surname> <given-names>S</given-names></string-name>, <string-name><surname>Liu</surname> <given-names>MY</given-names></string-name>, <string-name><surname>Yang</surname> <given-names>X</given-names></string-name>, <string-name><surname>Kautz</surname> <given-names>J</given-names></string-name></person-group>. <article-title>MoCoGAN: decomposing motion and content for video generation</article-title>. In: <conf-name>IEEE/CVF Conference on Computer Vision and Pattern Recognition</conf-name>; <year>2018 Jun 18&#x2013;23</year>; <publisher-loc>Salt Lake City, UT, USA</publisher-loc>: <publisher-name>IEEE</publisher-name>; 2018. p. <fpage>1526</fpage>&#x2013;<lpage>35</lpage>. doi:<pub-id pub-id-type="doi">10.1109/CVPR.2018.00165</pub-id>.</mixed-citation></ref>
<ref id="ref-46"><label>[46]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><surname>Rumelhart</surname> <given-names>DE</given-names></string-name>, <string-name><surname>Hinton</surname> <given-names>GE</given-names></string-name>, <string-name><surname>Williams</surname> <given-names>RJ</given-names></string-name></person-group>. <article-title>Learning representations by back-propagating errors</article-title>. <source>Nature</source>. <year>1986</year>;<volume>323</volume>(<issue>6088</issue>):<fpage>533</fpage>&#x2013;<lpage>6</lpage>. doi:<pub-id pub-id-type="doi">10.1038/323533a0</pub-id>.</mixed-citation></ref>
<ref id="ref-47"><label>[47]</label><mixed-citation publication-type="other"><person-group person-group-type="author"><string-name><surname>Tian</surname> <given-names>Y</given-names></string-name>, <string-name><surname>Ren</surname> <given-names>J</given-names></string-name>, <string-name><surname>Chai</surname> <given-names>M</given-names></string-name>, <string-name><surname>Olszewski</surname> <given-names>K</given-names></string-name>, <string-name><surname>Peng</surname> <given-names>X</given-names></string-name>, <string-name><surname>Metaxas</surname> <given-names>DN</given-names></string-name>, <etal>et al</etal></person-group>. <article-title>A good image generator is what you need for high-resolution video synthesis</article-title>. <comment>arXiv:2104.15069. 2021</comment>.</mixed-citation></ref>
<ref id="ref-48"><label>[48]</label><mixed-citation publication-type="other"><person-group person-group-type="author"><string-name><surname>Wang</surname> <given-names>X</given-names></string-name>, <string-name><surname>Qiao</surname> <given-names>T</given-names></string-name>, <string-name><surname>Zhu</surname> <given-names>J</given-names></string-name>, <string-name><surname>Hanjalic</surname> <given-names>A</given-names></string-name>, <string-name><surname>Scharenborg</surname> <given-names>O</given-names></string-name></person-group>. <article-title>S2IGAN: speech-to-image generation via adversarial learning</article-title>. <comment>arXiv:2005.06968. 2020</comment>.</mixed-citation></ref>
<ref id="ref-49"><label>[49]</label><mixed-citation publication-type="conf-proc"><person-group person-group-type="author"><string-name><surname>Wu</surname> <given-names>H</given-names></string-name>, <string-name><surname>Zheng</surname> <given-names>S</given-names></string-name>, <string-name><surname>Zhang</surname> <given-names>J</given-names></string-name>, <string-name><surname>Huang</surname> <given-names>K</given-names></string-name></person-group>. <article-title>GP-GAN: towards realistic high-resolution image blending</article-title>. In: <conf-name>Proceedings of the 27th ACM International Conference on Multimedia</conf-name>; <year>2019</year>; <publisher-loc>Nice France</publisher-loc>: <publisher-name>ACM</publisher-name>. p. <fpage>2487</fpage>&#x2013;<lpage>95</lpage>. doi:<pub-id pub-id-type="doi">10.1145/3343031.3350944</pub-id>.</mixed-citation></ref>
<ref id="ref-50"><label>[50]</label><mixed-citation publication-type="book"><person-group person-group-type="author"><string-name><surname>Wu</surname> <given-names>J</given-names></string-name>, <string-name><surname>Zhang</surname> <given-names>C</given-names></string-name>, <string-name><surname>Xue</surname> <given-names>T</given-names></string-name>, <string-name><surname>Freeman</surname> <given-names>WT</given-names></string-name>, <string-name><surname>Tenenbaum</surname> <given-names>JB</given-names></string-name></person-group>. <chapter-title>Learning a probabilistic latent space of object shapes via 3D generative-adversarial modeling</chapter-title>. In: <source>Advances in Neural Information Processing Systems (NeurIPS 2016)</source>; <year>2016</year>; <publisher-name>Curran Associates, Inc</publisher-name>. Vol. <volume>30</volume>.</mixed-citation></ref>
<ref id="ref-51"><label>[51]</label><mixed-citation publication-type="conf-proc"><person-group person-group-type="author"><string-name><surname>Xu</surname> <given-names>T</given-names></string-name>, <string-name><surname>Zhang</surname> <given-names>P</given-names></string-name>, <string-name><surname>Huang</surname> <given-names>Q</given-names></string-name>, <string-name><surname>Zhang</surname> <given-names>H</given-names></string-name>, <string-name><surname>Gan</surname> <given-names>Z</given-names></string-name>, <string-name><surname>Huang</surname> <given-names>X</given-names></string-name>, <etal>et al.</etal></person-group> <article-title>AttnGAN: fine-grained text to image generation with attentional generative adversarial networks</article-title>. In: <conf-name>2018 IEEE/CVF Conference on Computer Vision and Pattern Recognition</conf-name>; <year>2018 Jun 18&#x2013;23</year>; <publisher-loc>Salt Lake City, UT, USA</publisher-loc>: <publisher-name>IEEE</publisher-name>; 2018. p. <fpage>1316</fpage>&#x2013;<lpage>24</lpage>. doi:<pub-id pub-id-type="doi">10.1109/CVPR.2018.00143</pub-id>.</mixed-citation></ref>
<ref id="ref-52"><label>[52]</label><mixed-citation publication-type="conf-proc"><person-group person-group-type="author"><string-name><surname>Zhu</surname> <given-names>M</given-names></string-name>, <string-name><surname>Pan</surname> <given-names>P</given-names></string-name>, <string-name><surname>Chen</surname> <given-names>W</given-names></string-name>, <string-name><surname>Yang</surname> <given-names>Y</given-names></string-name></person-group>. <article-title>DM-GAN: dynamic memory generative adversarial networks for text-to-image synthesis</article-title>. In: <conf-name>2019 IEEE/CVF Conference on Computer Vision and Pattern Recognition (CVPR)</conf-name>; <year>2019 Jun 15&#x2013;20</year>; <publisher-loc>Long Beach, CA, USA</publisher-loc>: <publisher-name>IEEE</publisher-name>; 2019. p. <fpage>5802</fpage>&#x2013;<lpage>10</lpage>. doi:<pub-id pub-id-type="doi">10.1109/CVPR.2019.00595</pub-id>.</mixed-citation></ref>
<ref id="ref-53"><label>[53]</label><mixed-citation publication-type="conf-proc"><person-group person-group-type="author"><string-name><surname>Zhu</surname> <given-names>JY</given-names></string-name>, <string-name><surname>Park</surname> <given-names>T</given-names></string-name>, <string-name><surname>Isola</surname> <given-names>P</given-names></string-name>, <string-name><surname>Efros</surname> <given-names>AA</given-names></string-name></person-group>. <article-title>Unpaired image-to-image translation using cycle-consistent adversarial networks</article-title>. In: <conf-name>2017 IEEE International Conference on Computer Vision (ICCV)</conf-name>; <year>2017 Oct 22&#x2013;29</year>; <publisher-loc>Venice, Italy</publisher-loc>: <publisher-name>IEEE</publisher-name>; 2017. p. <fpage>2223</fpage>&#x2013;<lpage>32</lpage>. doi:<pub-id pub-id-type="doi">10.1109/ICCV.2017.244</pub-id>.</mixed-citation></ref>
<ref id="ref-54"><label>[54]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><surname>Borji</surname> <given-names>A</given-names></string-name></person-group>. <article-title>Pros and cons of GAN evaluation measures: new developments</article-title>. <source>Comput Vis Image Underst</source>. <year>2022</year>;<volume>215</volume>(<issue>4</issue>):<fpage>103329</fpage>. doi:<pub-id pub-id-type="doi">10.1016/j.cviu.2021.103329</pub-id>.</mixed-citation></ref>
<ref id="ref-55"><label>[55]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><surname>Figueira</surname> <given-names>A</given-names></string-name>, <string-name><surname>Vaz</surname> <given-names>B</given-names></string-name></person-group>. <article-title>Survey on synthetic data generation, evaluation methods and GANs</article-title>. <source>Mathematics</source>. <year>2022</year>;<volume>10</volume>(<issue>15</issue>):<fpage>2733</fpage>. doi:<pub-id pub-id-type="doi">10.3390/math10152733</pub-id>.</mixed-citation></ref>
</ref-list>
</back></article>