{
  "updated": "2026-10-05",
  "canonical_origin": "https://ghazanfarali.com",
  "papers": [
    {
      "slug": "congrets",
      "title": "ConGRets: Contrastive Learning for Scalable, Low-latency Co-speech Gesture Generation with Zero-shot Speaker Style Adaptation",
      "short_title": "ConGRets",
      "year": 2026,
      "type": "manuscript",
      "venue": "Manuscript under review",
      "status": "under review",
      "authors": [],
      "image": "assets/works/congrets.webp",
      "tagline": "Recorded gesture retrieval combines spoken content with motion-derived speaker style.",
      "summary": "ConGRets aligns text and motion-derived speaker style with recorded gesture units. Reference movement from an unfamiliar speaker supplies style context; inference retrieves existing motion without decoding new frames.",
      "pipeline": [
        "Text + reference motion",
        "Content/style contrastive retrieval",
        "Recorded gesture unit"
      ],
      "facts": {
        "Input": "Text and reference motion for speaker-style adaptation",
        "Method": "Contrastive content/style alignment and speaker-specific motion retrieval",
        "Output": "Retrieved co-speech gesture units",
        "Data and scope": "BEAT and curated motion libraries described in the manuscript",
        "Limitations": "Manuscript under review; retrieval is bounded by the recorded motion bank and available style context."
      },
      "related": [
        "ridge",
        "multilingual-gesture"
      ],
      "local_source_pdf": "assets/papers/congrets-manuscript.pdf",
      "sources": [
        "assets/papers/congrets-manuscript.pdf"
      ],
      "artifact_status": "Independent implementation of the paper’s core ideas, with setup instructions and data preparation documented in the repository README. The institute’s original source, datasets and trained models are not distributed.",
      "public_navigation": false,
      "code": "https://github.com/ghazanPK/congrets",
      "method_figure": {
        "image": "assets/research/congrets/method.png",
        "alt": "Method diagram from Figure 3 of the congrets paper",
        "caption": "Method diagram from the paper · Figure 3, PDF page 5.",
        "width": 1566,
        "height": 860,
        "responsive": [
          {
            "image": "assets/research/congrets/method-640.webp",
            "width": 640,
            "height": 351
          },
          {
            "image": "assets/research/congrets/method-1280.webp",
            "width": 1280,
            "height": 703
          },
          {
            "image": "assets/research/congrets/method-1566.webp",
            "width": 1566,
            "height": 860
          }
        ]
      }
    },
    {
      "year": 2026,
      "type": "manuscript",
      "title": "Lightweight Speech-Conditioned Upper-Face Animation for Virtual Agents via Emotion–Liveness Composition",
      "venue": "Manuscript in final review",
      "slug": "upper-face-animation",
      "short_title": "Upper-face animation",
      "image": "assets/works/upper-face.webp",
      "status": "in final review",
      "authors": [
        {
          "given": "Hwang Youn",
          "family": "Kim"
        },
        {
          "given": "Ghazanfar",
          "family": "Ali"
        },
        {
          "given": "Jeongha",
          "family": "Lee"
        },
        {
          "given": "Jae-In",
          "family": "Hwang"
        }
      ],
      "sources": [
        "assets/papers/upper-face-animation.pdf",
        "assets/papers/upper-face-animation.pdf"
      ],
      "intended_venue": "IEEE Access",
      "status_source": "Author confirmation, 2026-10-03",
      "tagline": "Speech, emotion, and subtle liveness cues compose efficient upper-face motion.",
      "summary": "The manuscript separates expressive upper-face animation from small liveness movements. A lightweight regressor and a motion prior produce nine eyebrow and eye-region blendshape channels; an external lip-sync system handles the mouth.",
      "pipeline": [
        "Speech and emotion",
        "Emotion + liveness",
        "Upper-face blendshapes"
      ],
      "facts": {
        "Input": "Speech acoustics, emotion category and intensity, speaker style",
        "Output": "Nine upper-face blendshape channels; lip motion is external",
        "Method": "Emotion generator plus separately trained liveness prior",
        "Data and scope": "Speech and facial-motion training data described in the manuscript",
        "Evaluation": "Manuscript reports 4.18M parameters, 13.16 ms CPU inference for about 3 seconds of audio, and a 36-person study",
        "Limitations": "Results are from a manuscript in final review; they are not published results. The system does not generate full-face motion."
      },
      "abstract": "Generating conversational facial expressions in real time still faces significant limitations, and most studies focus on lip-sync animation with limited attention to upper-face movements such as eyebrow and eye-region motion. Nevertheless, generating detailed facial movements using deep learning models requires considerable computational time and cost. In this paper, we propose a hybrid upperface animation framework that can be applied to various virtual agents with near-real-time computational performance. Unlike full-face facial-animation models, the proposed model directly generates only nine upper-face blendshape channels corresponding to eyebrow and eye-region movements, while lip motion is generated by an external real-time lip-sync module. The proposed framework consists of an emotionconditioned upper-face generator and a liveness generator. Emotional upper-face animation is generated by a lightweight Transformer-based regressor conditioned on acoustic features, emotion category, emotion intensity, and speaker style, while a separately trained generative motion prior uses speech features as weak temporal and prosodic cues to produce natural liveness variations. The proposed generators contain4.18M parameters in total, approximately1/30the number of parameters of EmoFace and1/150that of EmoTalk, and achieve a CPU inference time of13.16ms for approximately 3-s audio clips. Liveness-motion fusion increased the motion amplitude ratio from0.724to0.997. This indicates that the proposed method can generate upper-face motion with an amplitude close to that of real facial motion. Furthermore, in a user study with 36 participants, the proposed method achieved the highest mean Animacy rating and showed no significant post-hoc differences from EmoFace.",
      "abstract_source": "assets/papers/upper-face-animation.pdf",
      "artifact_status": "Independent implementation of the paper’s core ideas, with setup instructions and data preparation documented in the repository README. The institute’s original source, datasets and trained models are not distributed.",
      "related": [
        "lufa"
      ],
      "pdf_delivery": "external publisher or preprint archive",
      "local_source_pdf": "assets/papers/upper-face-animation.pdf",
      "availability_note": "An external manuscript archive link will be added when the public deposit is available.",
      "public_navigation": false,
      "code": "https://github.com/ghazanPK/upper-face-animation",
      "method_figure": {
        "image": "assets/research/upper-face-animation/method.png",
        "alt": "Method diagram from Figure 1 of the upper-face-animation paper",
        "caption": "Method diagram from the paper · Figure 1, PDF page 3.",
        "width": 1550,
        "height": 709,
        "responsive": [
          {
            "image": "assets/research/upper-face-animation/method-640.webp",
            "width": 640,
            "height": 293
          },
          {
            "image": "assets/research/upper-face-animation/method-1280.webp",
            "width": 1280,
            "height": 585
          },
          {
            "image": "assets/research/upper-face-animation/method-1550.webp",
            "width": 1550,
            "height": 709
          }
        ]
      }
    },
    {
      "year": 2026,
      "type": "conference paper",
      "title": "Through Van Gogh’s Eyes: Global Style Transfer with Diffusion Model",
      "venue": "ECCV",
      "slug": "global-style-transfer",
      "short_title": "Through Van Gogh’s Eyes",
      "image": "assets/works/van-gogh-style.webp",
      "status": "published",
      "authors": [
        {
          "given": "Jeongha",
          "family": "Lee"
        },
        {
          "given": "Yujin",
          "family": "Kim"
        },
        {
          "given": "Ghazanfar",
          "family": "Ali"
        },
        {
          "given": "Suhyun",
          "family": "Kim"
        },
        {
          "given": "Jae-In",
          "family": "Hwang"
        }
      ],
      "sources": [
        "https://api.crossref.org/works/10.1007/978-3-032-37092-1_32",
        "https://doi.org/10.1007/978-3-032-37092-1_32",
        "assets/papers/global-style-transfer.pdf"
      ],
      "doi": "10.1007/978-3-032-37092-1_32",
      "publication_date": "2026-09-11",
      "page": "571-588",
      "tagline": "A broader artist style distribution guides diffusion image synthesis.",
      "summary": "Global Style Transfer learns visual characteristics from a collection of an artist’s works rather than relying on one reference painting or a style prompt. Content Alignment Guidance preserves the content image’s semantic structure while allowing artistic deformation.",
      "pipeline": [
        "Content + artist corpus",
        "Style + content guidance",
        "Stylized image"
      ],
      "facts": {
        "Input": "Content image and a target artist’s artwork collection",
        "Output": "An image combining source content with artist-level style",
        "Method": "Global Style Transfer in diffusion h-space; Content Alignment Guidance",
        "Data and scope": "WikiArt artwork collections",
        "Evaluation": "Comparison with vanilla diffusion on style consistency, content alignment, and stylistic bias",
        "Limitations": "A collection-level style representation does not establish reproduction of an artist’s intent; results depend on the artwork corpus and diffusion model."
      },
      "abstract": "Diffusion models have achieved strong performance in artis-006 tic image synthesis, yet they still suffer from stylistic bias: when instruct-007 ing the diffusion model to create ‘Van Gogh-style’ images, we observed008 that the model tends to repeatedly generate textures and compositions009 characteristic of a narrow subset of iconic works. This bias limits the010 model’s ability to fully represent an artist’s stylistic diversity. To ad-011 dress this, we introduce Global Style Transfer (GST), a text-independent012 framework that learns an artist’s unified style distribution by training013 global visual statistics from hundreds of artworks. GST guides the diffu-014 sion process in the intermediate feature space (h-space), enabling gener-015 ation that reflects an artist’s global stylistic spectrum rather than relying016 on prompt-specific cues or memorized exemplars. In addition, we propose017 Content Alignment Guidance (CAG), a training-free guidance that aligns018 the semantic structure of a given content image while permitting flex-019 ible, style-based deformation. CAG preserves content identity without020 constraining artistic variation, allowing structural reinterpretations that021 naturally arise in artistic expression. Experiments on WikiArt bench-022 marks demonstrate that our method produces images with improved023 style consistency, reduced prompt-induced bias, and greater fidelity to024 global artistic semantics compared to the vanilla diffusion model. Our025 findings establish GST as a new direction for bias-robust artistic image026 generation.027",
      "abstract_source": "assets/papers/global-style-transfer.pdf",
      "artifact_status": "The original implementation is held by the institute. A separate public educational implementation is planned. No repository or trained weights are linked from this page yet.",
      "related": [],
      "preprint": "https://arxiv.org/abs/2608.11546",
      "pdf_delivery": "external publisher or preprint archive",
      "local_source_pdf": "assets/papers/global-style-transfer.pdf",
      "public_navigation": true,
      "navigation_title": "Through Van Gogh’s Eyes: Global Style Transfer"
    },
    {
      "year": 2025,
      "type": "journal article",
      "title": "Expanding Multilingual Co-Speech Interaction: The Impact of Enhanced Gesture Units in Text-to-Gesture Synthesis for Digital Humans",
      "venue": "IEEE Access",
      "role": "first author",
      "slug": "multilingual-gesture",
      "short_title": "Multilingual GestureCLR",
      "image": "assets/works/gestureclr-multilingual.webp",
      "status": "published",
      "authors": [
        {
          "given": "Ghazanfar",
          "family": "Ali"
        },
        {
          "given": "Woojoo",
          "family": "Kim"
        },
        {
          "given": "Muhammad Shahid",
          "family": "Anwar"
        },
        {
          "given": "Jae-In",
          "family": "Hwang"
        },
        {
          "given": "Ahyoung",
          "family": "Choi"
        }
      ],
      "sources": [
        "https://api.crossref.org/works/10.1109/access.2025.3596328",
        "https://doi.org/10.1109/access.2025.3596328",
        "assets/papers/multilingual-gesture.pdf"
      ],
      "doi": "10.1109/access.2025.3596328",
      "publication_date": "2025",
      "volume": "13",
      "page": "145144-145157",
      "tagline": "More gesture variety supports multilingual digital-human interaction.",
      "summary": "The system extracts text and 2D pose from English monologue videos, matches those poses to captured 3D gesture units with GestureCLR, and builds a text-to-gesture rule base. Non-English input is translated into English before gesture retrieval. The study compares no gestures, a small library, and the expanded library.",
      "pipeline": [
        "Video + captured motion",
        "GestureCLR rule-map",
        "Text-driven gesture retrieval"
      ],
      "facts": {
        "Input": "Text; non-English text translated into English",
        "Output": "Retrieved 3D co-speech gesture units for a digital human",
        "Method": "Contrastive 2D-to-3D matching offline; rule-based retrieval at runtime",
        "Data and scope": "English monologue videos; Korean-speaker motion capture; 2,035 units and 210,000 rules",
        "Evaluation": "51-participant study of gesture diversity and translation; reported high-noise matching improvement",
        "Limitations": "Multilingual support uses translation and an English rule base. The study does not establish equivalence across all languages or cultures."
      },
      "abstract": "In this study, we explore the effects of co-speech gesture generation on user experience in 3D digital human interaction by testing two key hypotheses. The first hypothesis posits that increasing the number of gestures enhances the user experience across criteria such as naturalness, human-likeness, temporal consistency, semantic consistency, and social presence. The second hypothesis suggests that language translation does not degrade the user experience across these criteria. To explore these hypotheses, we investigated three conditions using a digital human: voice only with no gestures, limited(56 gestures) cospeech gestures, and full system functionality with over 2000 unique gestures. For the second hypothesis, we used language translation to provide multilingual support, retrieving gestures from an English rule base. We obtained text and pose from English videos and matched the pose with gesture units derived from Korean speakers’ motion-capture sequences, enhancing a comprehensive rule base that we used for gesture retrieval for given text input. We used translation of non-English input language to English for text matching. Our novel method utilizes an improved pipeline to extract text, 2D pose data, and 3D gesture units. Incorporating a cutting-edge gesture-pose matching model with deep contrastive learning, we retrieved gestures from a comprehensive rule base containing 210,000 rules. This approach optimizes alignment and generates realistic, semantically consistent co-speech gestures adaptable to various languages. A comprehensive user study evaluated our hypotheses. The results underscored the positive impact of diverse gestures, supporting the first hypothesis. Additionally, multilingual capabilities did not degrade the user experience, confirming the second hypothesis. Highlighting the scalability and flexibility of our method, this study provides valuable insights into cross-lingual data and expert systems for gesture generation, contributing significantly to more engaging and immersive digital human interactions and the broader field of human-computer interaction.",
      "abstract_source": "assets/papers/multilingual-gesture.pdf",
      "artifact_status": "Independent implementation of the paper’s core ideas, with setup instructions and data preparation documented in the repository README. The institute’s original source, datasets and trained models are not distributed.",
      "related": [
        "ridge",
        "wild-pose-matching",
        "automatic-text-to-gesture"
      ],
      "preprint": "https://doi.org/10.21203/rs.3.rs-3350470/v1",
      "pdf_delivery": "external publisher or preprint archive",
      "local_source_pdf": "assets/papers/multilingual-gesture.pdf",
      "public_navigation": true,
      "code": "https://github.com/ghazanPK/multilingual-gesture",
      "method_figure": {
        "image": "assets/research/multilingual-gesture/graphical-abstract.png",
        "alt": "Graphical abstract: GestureCLR rule construction and translation-based multilingual gesture retrieval",
        "caption": "Graphical abstract diagram. GestureCLR expands a clustered gesture library for translation-based multilingual retrieval.",
        "width": 1536,
        "height": 1024,
        "responsive": [
          {
            "image": "assets/research/multilingual-gesture/graphical-abstract-640.webp",
            "width": 640,
            "height": 427
          },
          {
            "image": "assets/research/multilingual-gesture/graphical-abstract-1280.webp",
            "width": 1280,
            "height": 853
          },
          {
            "image": "assets/research/multilingual-gesture/graphical-abstract-1536.webp",
            "width": 1536,
            "height": 1024
          }
        ]
      },
      "navigation_title": "Multilingual Co-Speech Gesture Synthesis"
    },
    {
      "year": 2025,
      "type": "journal article",
      "title": "RIDGE: Rule‐Infused Deep Learning for Realistic Co‐Speech Gesture Generation",
      "venue": "Computer Animation and Virtual Worlds",
      "role": "first author",
      "slug": "ridge",
      "short_title": "RIDGE",
      "image": "assets/works/ridge.webp",
      "status": "published",
      "authors": [
        {
          "given": "Ghazanfar",
          "family": "Ali"
        },
        {
          "given": "HwangYoun",
          "family": "Kim"
        },
        {
          "given": "Jae‐In",
          "family": "Hwang"
        }
      ],
      "sources": [
        "https://api.crossref.org/works/10.1002/cav.70034",
        "https://doi.org/10.1002/cav.70034",
        "assets/papers/ridge.pdf"
      ],
      "doi": "10.1002/cav.70034",
      "publication_date": "2025-08-07",
      "volume": "36",
      "issue": "4",
      "article_number": "e70034",
      "tagline": "High-confidence rules and learned similarity retrieve semantically aligned gestures.",
      "summary": "RIDGE first searches a motion-derived rule base enriched with language-model assistance. When a rule does not meet the confidence threshold, a contrastively trained text–motion embedding retrieves an appropriate recorded gesture. Both paths use existing animation segments; direct decoding into new motion frames is described as future work.",
      "pipeline": [
        "Text query",
        "Rule match or learned fallback",
        "Recorded gesture clip"
      ],
      "facts": {
        "Input": "Text",
        "Output": "Retrieved recorded gesture clips",
        "Method": "Rule retrieval with confidence gating; contrastive text–motion retrieval fallback",
        "Data and scope": "BEAT co-speech motion and in-the-wild video data",
        "Evaluation": "Gesture Cluster Affinity: RIDGE 0.73, rule baseline 0.60, end-to-end baseline 0.52, ground truth 0.90",
        "Limitations": "Direct latent-to-motion decoding is outside the study. Retrieved motion inherits source quality, including finger artifacts."
      },
      "abstract": "Co-speech gestures are essential for natural human communication, yet existing synthesis methods fall short in delivering semantically aligned and contextually appropriate motions. In this paper, we present RIDGE, a hybrid system that combines rule-based and deep learning approaches to generate realistic gestures for virtual avatars and human-computer interaction. RIDGE employs a high-fidelity rule base generated from motion capture data with the assistance of large language models, to select reliable gesture mappings. When a high-confidence match is not available, a contrastively trained deep learning model steps in to produce semantically appropriate gestures. Evaluated using a novel Gesture Cluster Affinity (GCA) metric, our system outperforms existing baselines, achieving a GCA score of 0.73 compared to rule-based baseline 0.6 and end-toend: 0.52, while ground truth score was 0.90. Detailed analyses of system architecture, data preprocessing, and evaluation methodologies demonstrate RIDGE’s potential to enhance gesture synthesis. Project Url: https://www. mrlab.co.kr/research/ridge",
      "abstract_source": "assets/papers/ridge.pdf",
      "artifact_status": "Independent implementation of the paper’s core ideas, with setup instructions and data preparation documented in the repository README. The institute’s original source, datasets and trained models are not distributed.",
      "related": [
        "multilingual-gesture",
        "wild-pose-matching",
        "automatic-text-to-gesture"
      ],
      "pdf_delivery": "external publisher or preprint archive",
      "local_source_pdf": "assets/papers/ridge.pdf",
      "public_navigation": true,
      "code": "https://github.com/ghazanPK/ridge",
      "method_figure": {
        "image": "assets/research/ridge/ridge-system.png",
        "alt": "RIDGE system architecture: rule-base construction, contrastive representation learning, and threshold-gated hybrid gesture retrieval",
        "caption": "Graphical abstract diagram. Confidence-gated rules and learned similarity retrieve recorded gesture clips.",
        "width": 1536,
        "height": 1024,
        "responsive": [
          {
            "image": "assets/research/ridge/ridge-system-640.webp",
            "width": 640,
            "height": 427
          },
          {
            "image": "assets/research/ridge/ridge-system-1280.webp",
            "width": 1280,
            "height": 853
          },
          {
            "image": "assets/research/ridge/ridge-system-1536.webp",
            "width": 1536,
            "height": 1024
          }
        ]
      },
      "navigation_title": "RIDGE: Realistic Co-Speech Gesture Generation"
    },
    {
      "year": 2025,
      "type": "journal article",
      "title": "A Retrieval‐Augmented Generation System for Accurate and Contextual Historical Analysis: AI‐Agent for the Annals of the Joseon Dynasty",
      "venue": "Computer Animation and Virtual Worlds",
      "slug": "joseon-rag-journal",
      "short_title": "Joseon RAG journal",
      "image": "assets/works/joseon-rag.webp",
      "status": "published",
      "authors": [
        {
          "given": "Jeong Ha",
          "family": "Lee"
        },
        {
          "given": "Ghazanfar",
          "family": "Ali"
        },
        {
          "given": "Jae‐In",
          "family": "Hwang"
        }
      ],
      "sources": [
        "https://api.crossref.org/works/10.1002/cav.70048",
        "https://doi.org/10.1002/cav.70048",
        "assets/papers/joseon-rag-journal.pdf"
      ],
      "doi": "10.1002/cav.70048",
      "publication_date": "2025-07-20",
      "volume": "36",
      "issue": "4",
      "article_number": "e70048",
      "tagline": "Article-aware retrieval grounds answers in the Annals of the Joseon Dynasty.",
      "summary": "The agent preserves historical article boundaries, uses date metadata and query refinement to retrieve evidence, and supplies that evidence to a language model. Its purpose is to support factual answers and contextual interpretation with traceable sources.",
      "pipeline": [
        "Historical question",
        "Date-aware evidence retrieval",
        "Answer + source citations"
      ],
      "facts": {
        "Input": "A historical question and the Annals corpus",
        "Output": "Source-grounded answers and contextual analysis",
        "Method": "Article-aware chunking, date-aware retrieval, query refinement, and LLM response generation",
        "Data and scope": "Annals of the Joseon Dynasty",
        "Evaluation": "Paper reports improvements of approximately 23–50 points on its 100-point evaluation scale over the compared systems",
        "Limitations": "Evidence comes from the paper’s historical task and benchmark; it does not guarantee factual correctness for arbitrary questions."
      },
      "abstract": "In this paper, we propose an AI-agent that integrates a large language model(LLM) with a Retrieval-Augmented Generation(RAG) system to deliver reliable historical information from the Annals of the Joseon Dynasty through both objective facts and contextual analysis, achieving significant performance improvements over existing models. In order for an AI-agent using the Annals of the Joseon Dynasty to deliver reliable historical information, clear source citations and systematic analysis are essential. The Annals, an official record spanning 472 years (1392–1897), offer a dense, chronological account of daily events and state administration that shaped Korea’s cultural, political, and social foundations. We propose integrating a LLM with a RAG system to generate highly accurate responses based on this extensive dataset. This approach provides both objective information about historical figures and events from specific periods and subjective contextual analysis of the era, helping users gain a broader understanding. Our experiments demonstrate improvements of approximately 23 to 50 points on a 100-point scale compared to the GPT-4o and OpenAI AIAssistant v2 models.",
      "abstract_source": "assets/papers/joseon-rag-journal.pdf",
      "artifact_status": "Independent implementation of the paper’s core ideas, with setup instructions and data preparation documented in the repository README. The institute’s original source, datasets and trained models are not distributed.",
      "related": [
        "joseon-rag-ismar"
      ],
      "pdf_delivery": "external publisher or preprint archive",
      "local_source_pdf": "assets/papers/joseon-rag-journal.pdf",
      "public_navigation": true,
      "code": "https://github.com/ghazanPK/joseon-rag-journal",
      "method_figure": {
        "image": "assets/research/joseon-rag-journal/method.png",
        "alt": "Method diagram from Figure 1 of the joseon-rag-journal paper",
        "caption": "Method diagram from the paper · Figure 1, PDF page 2.",
        "width": 1360,
        "height": 658,
        "responsive": [
          {
            "image": "assets/research/joseon-rag-journal/method-640.webp",
            "width": 640,
            "height": 310
          },
          {
            "image": "assets/research/joseon-rag-journal/method-1280.webp",
            "width": 1280,
            "height": 619
          },
          {
            "image": "assets/research/joseon-rag-journal/method-1360.webp",
            "width": 1360,
            "height": 658
          }
        ]
      },
      "navigation_title": "Historical Analysis of the Joseon Dynasty with RAG"
    },
    {
      "year": 2025,
      "type": "journal article",
      "title": "ASAP for multi-outputs: auto-generating storyboard and pre-visualization with virtual actors based on screenplay",
      "venue": "Multimedia Tools and Applications",
      "role": "equal contribution",
      "slug": "asap-journal",
      "short_title": "ASAP journal",
      "image": "assets/works/asap.webp",
      "status": "published",
      "authors": [
        {
          "given": "Hanseob",
          "family": "Kim"
        },
        {
          "given": "Ghazanfar",
          "family": "Ali"
        },
        {
          "given": "Bin",
          "family": "Han"
        },
        {
          "given": "Hwang Youn",
          "family": "Kim"
        },
        {
          "given": "Jieun",
          "family": "Kim"
        },
        {
          "given": "Hyemin",
          "family": "Shin"
        },
        {
          "given": "Gerard Jounghyun",
          "family": "Kim"
        },
        {
          "given": "Jae-In",
          "family": "Hwang"
        }
      ],
      "sources": [
        "https://api.crossref.org/works/10.1007/s11042-024-19904-3",
        "https://doi.org/10.1007/s11042-024-19904-3",
        "assets/papers/asap-journal.pdf"
      ],
      "doi": "10.1007/s11042-024-19904-3",
      "publication_date": "2024-08-03",
      "volume": "84",
      "issue": "20",
      "page": "22377-22400",
      "tagline": "A screenplay becomes storyboards, animated previsualization, and immersive scenes.",
      "summary": "ASAP parses screenplay structure into characters, dialogue, actions, and emotions. Its modules select virtual actors and compose speech gestures, physical actions, facial expressions, and scene outputs. The journal expands the earlier ISMAR system and SIGGRAPH Asia live demonstration.",
      "pipeline": [
        "Screenplay",
        "Characters + actions + emotion",
        "Storyboard / 3D / VR"
      ],
      "facts": {
        "Input": "A structured screenplay",
        "Output": "2D storyboards, animated 3D previews, and immersive VR scenes",
        "Method": "Screenplay parsing and coordinated character-behavior modules",
        "Data and scope": "Screenplay examples and action-sentence evaluation described in the paper",
        "Evaluation": "Reported top-1 action accuracy: 93% on simple sentences and 87% on complex sentences",
        "Limitations": "Animation coverage and scene composition depend on the available characters, actions, props, and environment assets."
      },
      "abstract": "One of the pressing desires of content creators is to be able to visualize how their characters will look in a scene as soon as possible. In the early stages of film production, this desire can be partly achieved by the computer graphics-based process known as Pre-visualization (Previz). However, traditional previz necessitates a high level of expertise and is also time-consuming. This paper introduces the ASAP system, an automated tool that creates pre-visualized animations and storyboards by generating virtual character behavior/animations based on understanding the screenplay. The ASAP system parses the user-written screenplay to extract data, including character names, dialogue, actions, and emotions. This extracted data is then passed to the respective modules, which select virtual characters and automatically generate their speaking gestures, physical movements, and expressive behaviors. We demonstrate the system’s fidelity by presenting multiple outputs, including a 2D storyboard, a 3D preview, and a VR-based immersive scenario, along with simulations of potential use cases. The ASAP system can streamline pre-visualization tasks in the pre-production phase and has the potential to be widely adopted by the film industry.",
      "abstract_source": "assets/papers/asap-journal.pdf",
      "artifact_status": "Independent implementation of the paper’s core ideas, with setup instructions and data preparation documented in the repository README. The institute’s original source, datasets and trained models are not distributed.",
      "related": [
        "asap-live",
        "asap-ismar"
      ],
      "related_video": {
        "id": "omdEg7Ro_bU",
        "label": "Related ASAP system: SIGGRAPH Asia 2022 live presentation",
        "source": "https://sites.google.com/view/mrlabkist/video-demo"
      },
      "pdf_delivery": "external publisher or preprint archive",
      "local_source_pdf": "assets/papers/asap-journal.pdf",
      "public_navigation": true,
      "code": "https://github.com/ghazanPK/asap-journal",
      "method_figure": {
        "image": "assets/research/asap-journal/method.png",
        "alt": "Original ASAP preparation and runtime architecture, Figure 3",
        "caption": "Method diagram from the paper · Figure 3, PDF page 8.",
        "width": 1401,
        "height": 724,
        "responsive": [
          {
            "image": "assets/research/asap-journal/method-640.webp",
            "width": 640,
            "height": 331
          },
          {
            "image": "assets/research/asap-journal/method-1280.webp",
            "width": 1280,
            "height": 661
          },
          {
            "image": "assets/research/asap-journal/method-1401.webp",
            "width": 1401,
            "height": 724
          }
        ]
      },
      "navigation_title": "ASAP: Storyboards and 3D Previsualization"
    },
    {
      "year": 2025,
      "type": "adjunct paper",
      "title": "RAG based AI-Agent for Contextualized Analysis of High-Density Historical Records: Application to the Annals of the Joseon Dynasty",
      "venue": "IEEE ISMAR-Adjunct",
      "slug": "joseon-rag-ismar",
      "short_title": "Joseon RAG adjunct",
      "image": "assets/works/joseon-rag.webp",
      "status": "published",
      "authors": [
        {
          "given": "Jeongha",
          "family": "Lee"
        },
        {
          "given": "Ghazanfar",
          "family": "Ali"
        },
        {
          "given": "Jae-In",
          "family": "Hwang"
        }
      ],
      "sources": [
        "https://api.crossref.org/works/10.1109/ismar-adjunct68609.2025.00243",
        "https://doi.org/10.1109/ismar-adjunct68609.2025.00243",
        "assets/papers/joseon-rag-ismar.pdf"
      ],
      "doi": "10.1109/ismar-adjunct68609.2025.00243",
      "publication_date": "2025-10-08",
      "page": "893-894",
      "tagline": "An embodied agent makes historical source retrieval conversational.",
      "summary": "This adjunct paper integrates a retrieval-augmented language model with a Unity-based 3D agent. Questions produce grounded historical answers delivered through synchronized voice and body motion. It is a separate publication from the related journal article.",
      "pipeline": [
        "User question",
        "Historical RAG",
        "Embodied answer"
      ],
      "facts": {
        "Input": "Voice or text questions about the Annals",
        "Output": "Historical answers with voice and contextual body animation",
        "Method": "Historical RAG pipeline integrated with a Unity agent",
        "Data and scope": "Annals corpus of 49,646,667 characters; 30 benchmark questions",
        "Evaluation": "Comparison on factual accuracy, reliability, and reasonableness against the paper’s GPT-4o and AI-Assistant v2 baselines",
        "Limitations": "A 30-question historical benchmark is limited in scope; the journal’s measurements should not be treated as measurements of this adjunct paper."
      },
      "abstract": "The field of digital heritage has increasingly focused on digitizing and reinterpreting historical materials to preserve and transmit cultural assets. Accurately conveying the content of historical records is critical for both academic research and education. Traditional digital archiving and retrieval systems have commonly relied on keyword-based searches or simple text matching, resulting in limited contextual understanding, insufficient source citation, and inadequate alignment with user intent. To address these challenges, this paper proposes a novel methodology that integrates a large language model (LLM) with a Retrieval-Augmented Generation (RAG) framework for high-density historical records such as the Annals of the Joseon Dynasty (49,646,667 characters, 13921910).Our system retrieves the most relevant historical sources, and delivers both objective facts and contextual analysis. Ultimately, we integrate this system with a Unity-based 3D AI-Agent, providing answers through synchronized voice output and body motion for an immersive interactive experience. By enabling natural, embodied interaction, this integration makes historical knowledge more accessible and engaging for users. Experimental results on a set of 30 benchmark questions demonstrate that our model outperforms both ChatGPT-4o and AI-Assistant v2 in Factual Accuracy, Reliability, and Reasonableness. This research expands the potential applications of digital heritage and lays the groundwork for broader integration with diverse historical data sources.",
      "abstract_source": "assets/papers/joseon-rag-ismar.pdf",
      "artifact_status": "Independent implementation of the paper’s core ideas, with setup instructions and data preparation documented in the repository README. The institute’s original source, datasets and trained models are not distributed.",
      "related": [
        "joseon-rag-journal"
      ],
      "pdf_delivery": "external publisher or preprint archive",
      "local_source_pdf": "assets/papers/joseon-rag-ismar.pdf",
      "public_navigation": true,
      "code": "https://github.com/ghazanPK/joseon-rag-ismar",
      "method_figure": {
        "image": "assets/research/joseon-rag-ismar/method.png",
        "alt": "Method diagram from Figure 2 of the joseon-rag-ismar paper",
        "caption": "Method diagram from the paper · Figure 2, PDF page 2.",
        "width": 1553,
        "height": 523,
        "responsive": [
          {
            "image": "assets/research/joseon-rag-ismar/method-640.webp",
            "width": 640,
            "height": 216
          },
          {
            "image": "assets/research/joseon-rag-ismar/method-1280.webp",
            "width": 1280,
            "height": 431
          },
          {
            "image": "assets/research/joseon-rag-ismar/method-1553.webp",
            "width": 1553,
            "height": 523
          }
        ]
      }
    },
    {
      "year": 2025,
      "type": "adjunct paper",
      "title": "LUFA: Lightweight Upper-Face Animation for VR/MR Avatars",
      "venue": "IEEE ISMAR-Adjunct",
      "slug": "lufa",
      "short_title": "LUFA",
      "image": "assets/works/upper-face.webp",
      "status": "published",
      "authors": [
        {
          "given": "Hwang Youn",
          "family": "Kim"
        },
        {
          "given": "Ghazanfar",
          "family": "Ali"
        },
        {
          "given": "Jae-In",
          "family": "Hwang"
        }
      ],
      "sources": [
        "https://api.crossref.org/works/10.1109/ismar-adjunct68609.2025.00217",
        "https://doi.org/10.1109/ismar-adjunct68609.2025.00217",
        "https://api.openalex.org/works/https://doi.org/10.1109/ismar-adjunct68609.2025.00217"
      ],
      "doi": "10.1109/ismar-adjunct68609.2025.00217",
      "publication_date": "2025-10-08",
      "page": "841-842",
      "tagline": "Lightweight upper-face animation for VR and MR avatars.",
      "summary": "LUFA encodes voice and text with fine-tuned Wav2Vec2.0 and BERT models. Reconstruction loss and contrastive learning align latent representations, which are used to retrieve facial-animation sequences for VR and MR avatars. The related journal manuscript studies a later emotion–liveness approach and is in final review.",
      "pipeline": [
        "Voice + text",
        "Aligned latent representations",
        "Facial animation retrieval"
      ],
      "facts": {
        "Input": "Voice and text",
        "Output": "Retrieved upper-face facial-animation sequences",
        "Method": "Wav2Vec2.0 and BERT encoders; reconstruction loss; contrastive latent alignment and retrieval",
        "Data and scope": "Conference abstract describes the representation and retrieval framework",
        "Evaluation": "Consult the full conference paper for evaluation details",
        "Limitations": "Retrieval selects existing animation sequences. The later emotion–liveness manuscript’s parameter counts and study results do not describe LUFA."
      },
      "artifact_status": "Independent implementation of the paper’s core ideas, with setup instructions and data preparation documented in the repository README. The institute’s original source, datasets and trained models are not distributed.",
      "related": [
        "upper-face-animation"
      ],
      "abstract": "For virtual agents, realistic co-speech facial expressions are essential to enhance naturalness. Rule-based methods lack diversity and temporal consistency in generating emotional expressions. Additionally, generating facial animation using large-scale generative models requires substantial computational resources, making real-time deployment challenging. In this paper, we propose Lightweight Upper-Face Animation for VR/MR Avatars (LUFA), a co-speech facial expression framework for generating real-time animations from voice and text inputs. We fine-tune Wav2Vec2.0 and BERT encoders using a reconstruction loss. Our framework treats their outputs as latent representations, aligns them through contrastive learning, and retrieves facial animation sequences based on these representations.",
      "abstract_source": "https://api.openalex.org/works/https://doi.org/10.1109/ismar-adjunct68609.2025.00217",
      "pdf_delivery": "external publisher or preprint archive",
      "public_navigation": true,
      "code": "https://github.com/ghazanPK/lufa",
      "method_figure": {
        "image": "assets/research/lufa/graphical-abstract.png",
        "alt": "Graphical abstract: aligned voice and text representations retrieve upper-face animation for VR and MR avatars",
        "caption": "Graphical abstract diagram. Aligned speech and text representations support recorded facial-motion retrieval.",
        "width": 1942,
        "height": 809,
        "responsive": [
          {
            "image": "assets/research/lufa/graphical-abstract-640.webp",
            "width": 640,
            "height": 267
          },
          {
            "image": "assets/research/lufa/graphical-abstract-1280.webp",
            "width": 1280,
            "height": 533
          },
          {
            "image": "assets/research/lufa/graphical-abstract-1920.webp",
            "width": 1920,
            "height": 800
          }
        ]
      }
    },
    {
      "year": 2024,
      "type": "journal article",
      "title": "Enhancing doctor‐patient communication in surgical explanations: Designing effective facial expressions and gestures for animated physician characters",
      "venue": "Computer Animation and Virtual Worlds",
      "award": "Best Paper Award, CASA 2024",
      "slug": "virtual-physician",
      "short_title": "Virtual physician",
      "image": "assets/works/virtual-physician.webp",
      "status": "published",
      "authors": [
        {
          "given": "Hwang Youn",
          "family": "Kim"
        },
        {
          "given": "Ghazanfar",
          "family": "Ali"
        },
        {
          "given": "Jae‐In",
          "family": "Hwang"
        }
      ],
      "sources": [
        "https://api.crossref.org/works/10.1002/cav.2236",
        "https://doi.org/10.1002/cav.2236",
        "assets/papers/virtual-physician.pdf"
      ],
      "doi": "10.1002/cav.2236",
      "publication_date": "2024-06-06",
      "volume": "35",
      "issue": "3",
      "article_number": "e2236",
      "tagline": "Expression and gesture support surgical explanations by animated physicians.",
      "summary": "The system lets clinicians prepare explanations with a virtual physician and lets patients ask follow-up questions. It combines grounded medical information with designed facial expression and co-speech gesture, and evaluates how users perceive the presentation.",
      "pipeline": [
        "Surgical content / question",
        "Grounded answer + behavior",
        "Animated explanation"
      ],
      "facts": {
        "Input": "Clinician-prepared surgical content and patient questions",
        "Output": "Animated surgical explanations and grounded follow-up responses",
        "Method": "Content preparation, grounded question answering, facial expression, and gesture",
        "Data and scope": "Surgical-explanation materials and study conditions described in the paper",
        "Evaluation": "113-participant study; reported answer F1 comparison 0.492 to 0.779",
        "Limitations": "This is communication research. Perceived social presence and answer scores do not demonstrate improved clinical outcomes or autonomous medical decision-making."
      },
      "abstract": "Paying close attention to facial expressions, gestures, and communication techniques is essential when creating animated physician characters that are realistic and captivating when describing surgical procedures. This paper emphasizes the integration of appropriate emotions, co-speech gestures when medical experts explain the medical procedure, and designing animated characters. We can achieve healthy doctor-patient relationships and improvement of patients’ understanding by depicting these components truthfully. We suggest two critical approaches to developing virtual medical experts by incorporating these elements. First, doctors can generate the contents of the surgical procedure with a virtual doctor. Second, patients can listen to the surgical procedure described by the virtual doctor and ask if they have any questions. Our system helps patients by considering their psychology and adding medical professionals’ opinions. These improvements ensure the animated virtual agent is comforting, reassuring, and emotionally supportive. Through a user study, we evaluated our hypothesis and gained insight into improvements.",
      "abstract_source": "assets/papers/virtual-physician.pdf",
      "artifact_status": "Independent implementation of the paper’s core ideas, with setup instructions and data preparation documented in the repository README. The institute’s original source, datasets and trained models are not distributed.",
      "related": [],
      "pdf_delivery": "external publisher or preprint archive",
      "local_source_pdf": "assets/papers/virtual-physician.pdf",
      "public_navigation": true,
      "code": "https://github.com/ghazanPK/virtual-physician",
      "method_figure": {
        "image": "assets/research/virtual-physician/graphical-abstract.png",
        "alt": "Graphical abstract: grounded surgical explanations, expressive avatar behavior and communication study findings",
        "caption": "Graphical abstract diagram. Grounded explanations and expressive behavior support virtual-physician communication.",
        "width": 1774,
        "height": 887,
        "responsive": [
          {
            "image": "assets/research/virtual-physician/graphical-abstract-640.webp",
            "width": 640,
            "height": 320
          },
          {
            "image": "assets/research/virtual-physician/graphical-abstract-1280.webp",
            "width": 1280,
            "height": 640
          },
          {
            "image": "assets/research/virtual-physician/graphical-abstract-1774.webp",
            "width": 1774,
            "height": 887
          }
        ]
      }
    },
    {
      "year": 2022,
      "type": "poster",
      "title": "Improving Co-speech gesture rule-map generation via wild pose matching with gesture units.",
      "venue": "SIGGRAPH Asia Posters",
      "role": "first author",
      "slug": "wild-pose-matching",
      "short_title": "Wild Pose Matching",
      "image": "assets/works/wild-pose.webp",
      "status": "published",
      "authors": [
        {
          "given": "Ghazanfar",
          "family": "Ali"
        },
        {
          "given": "Jae-In",
          "family": "Hwang"
        }
      ],
      "sources": [
        "https://api.crossref.org/works/10.1145/3550082.3564185",
        "https://doi.org/10.1145/3550082.3564185",
        "assets/papers/wild-pose-matching.pdf"
      ],
      "doi": "10.1145/3550082.3564185",
      "publication_date": "2022-12-13",
      "page": "1-2",
      "tagline": "Contrastive matching turns noisy video poses into a richer gesture rule map.",
      "summary": "The poster aligns 2D poses from public monologue videos with gesture units extracted from 3D motion capture. GestureCLR learns robust matching; K-Means clusters the units to support variety when retrieving gestures at runtime.",
      "pipeline": [
        "Text + noisy video pose",
        "GestureCLR matching",
        "Clustered gesture rules"
      ],
      "facts": {
        "Input": "Video-derived text and 2D pose; captured 3D motion",
        "Output": "Text-to-gesture rules and clustered gesture units",
        "Method": "Contrastive pose-to-unit matching and K-Means clustering",
        "Data and scope": "2,035 gesture units and 210,000 rules",
        "Evaluation": "Poster demonstrates the expanded gesture library and mapping pipeline",
        "Limitations": "This is a two-page poster; the later multilingual paper contains a separate user study and should be cited for that evidence."
      },
      "abstract": "In this poster, we present a method to generate co-speech textto-gesture mapping for 3D digital humans. We obtained text and 2D pose data from public monologue videos. Gesture units were obtained from motion capture sequences. The method works by matching 2D poses to 3D gesture units. We trained a model via contrastive learning to improve the matching of noisy pose sequences with gesture units. To ensure diverse gesture sequences at runtime, gesture units were clustered using K-Mean clustering. We incorporated 2035 gestures and 210k rules. Our method is highly adaptable and easy to control and use. Demo Video : https://youtu.be/QBtGdGE1Wgk",
      "abstract_source": "assets/papers/wild-pose-matching.pdf",
      "artifact_status": "Independent implementation of the paper’s core ideas, with setup instructions and data preparation documented in the repository README. The institute’s original source, datasets and trained models are not distributed.",
      "related": [
        "multilingual-gesture",
        "ridge",
        "automatic-text-to-gesture"
      ],
      "video": {
        "id": "QBtGdGE1Wgk",
        "label": "Paper presentation / demo",
        "source": "https://sites.google.com/view/mrlabkist/video-demo"
      },
      "pdf_delivery": "external publisher or preprint archive",
      "local_source_pdf": "assets/papers/wild-pose-matching.pdf",
      "public_navigation": true,
      "code": "https://github.com/ghazanPK/wild-pose-matching",
      "method_figure": {
        "image": "assets/research/wild-pose-matching/method.png",
        "alt": "Method diagram from Figure 1 of the wild-pose-matching paper",
        "caption": "Method diagram from the paper · Figure 1, PDF page 1.",
        "width": 1554,
        "height": 347,
        "responsive": [
          {
            "image": "assets/research/wild-pose-matching/method-640.webp",
            "width": 640,
            "height": 143
          },
          {
            "image": "assets/research/wild-pose-matching/method-1280.webp",
            "width": 1280,
            "height": 286
          },
          {
            "image": "assets/research/wild-pose-matching/method-1554.webp",
            "width": 1554,
            "height": 347
          }
        ]
      }
    },
    {
      "year": 2022,
      "type": "live demonstration",
      "title": "ASAP: Auto-generating Storyboard and Previz",
      "venue": "SIGGRAPH Asia Real-Time Live!",
      "slug": "asap-live",
      "short_title": "ASAP live demonstration",
      "image": "assets/works/asap.webp",
      "status": "published",
      "authors": [
        {
          "given": "Hanseob",
          "family": "Kim"
        },
        {
          "given": "Ghazanfar",
          "family": "Ali"
        },
        {
          "given": "Bin",
          "family": "Han"
        },
        {
          "given": "Hwangyoun",
          "family": "Kim"
        },
        {
          "given": "Jieun",
          "family": "Kim"
        },
        {
          "given": "Jae-In",
          "family": "Hwang"
        }
      ],
      "sources": [
        "https://api.crossref.org/works/10.1145/3550453.3570124",
        "https://doi.org/10.1145/3550453.3570124"
      ],
      "doi": "10.1145/3550453.3570124",
      "online_date": "2023-05-22",
      "publication_date": "2022-12-06",
      "page": "1-1",
      "tagline": "A live demonstration of screenplay-driven virtual actors.",
      "summary": "The SIGGRAPH Asia Real-Time Live! presentation shows ASAP turning screenplay text into virtual-actor scenes. Dialogue drives co-speech gesture, parentheticals supply emotional cues, and action paragraphs specify physical movements. The video is the live demonstration of this system.",
      "pipeline": [
        "Movie script",
        "Gesture + expression + action",
        "Live virtual-actor scene"
      ],
      "facts": {
        "Input": "Movie-script dialogue, parentheticals, and action paragraphs",
        "Output": "Virtual-actor scenes, gestures, facial expression, and body movements",
        "Method": "Script understanding and composition of animation modules",
        "Data and scope": "Demonstration scenarios",
        "Evaluation": "Real-Time Live! demonstration; no journal benchmark is attributed to this item",
        "Limitations": "This demonstration and its one-page publication are distinct from the ISMAR and journal papers."
      },
      "artifact_status": "Independent implementation of the paper’s core ideas, with setup instructions and data preparation documented in the repository README. The institute’s original source, datasets and trained models are not distributed.",
      "related": [
        "asap-journal",
        "asap-ismar"
      ],
      "external_pdf": "https://history.siggraph.org/wp-content/uploads/2025/09/2022SA_RealTimeLive_Kim_ASAP.pdf",
      "video": {
        "id": "omdEg7Ro_bU",
        "label": "SIGGRAPH Asia 2022 Real-Time Live! presentation",
        "source": "https://sites.google.com/view/mrlabkist/video-demo"
      },
      "abstract": "We present ASAP, a system that uses virtual humans to Automatically generate Storyboards And Pre-visualized scenes from movie scripts. In our ASAP system, virtual humans play the role of actors. To visualize the screenplay scene, our system understands the movie script, which is the text data, and then facilitates the automatic generation of the following virtual human’s non-/verbal behavior: (1) co-speech gesture, (2) facial expression, and (3) body movements. First of all, co-speech gestures are created from dialogue paragraphs using a text-to-gesture model trained with 2D videos and 3D motion-captured data. Next, for the facial expressions, we interpret the actors’ emotions in the parenthetical paragraphs and then adjust the virtual human’s face animation to reflect emotions such as anger and sadness. For body movements, our system extract action entities from action paragraphs (e.g., subject, target, and action) and then combine sets of animations to make animation sequences (e.g., a man’s act of sitting on a bed). As soon as possible, ASAP can reduce the amount of time, money, and labor-intensive work that needs to be done in the early stages of filmmaking.",
      "abstract_source": "https://history.siggraph.org/wp-content/uploads/2025/09/2022SA_RealTimeLive_Kim_ASAP.pdf",
      "pdf_delivery": "external publisher or preprint archive",
      "public_navigation": true,
      "code": "https://github.com/ghazanPK/asap-live",
      "method_figure": {
        "image": "assets/research/asap-live/method.svg",
        "alt": "Scientific method schematic for asap-live",
        "caption": "Graphical abstract diagram. Screenplay structure drives coordinated virtual-actor behavior and previsualization.",
        "width": 1500,
        "height": 610
      }
    },
    {
      "year": 2022,
      "type": "poster",
      "title": "No-code Digital Human for Conversational Behavior",
      "venue": "SIGGRAPH Asia Posters",
      "slug": "flow-human",
      "short_title": "Flow Human",
      "image": "assets/works/flow-human.webp",
      "status": "published",
      "authors": [
        {
          "given": "Hanseob",
          "family": "Kim"
        },
        {
          "given": "Jieun",
          "family": "Kim"
        },
        {
          "given": "Ghazanfar",
          "family": "Ali"
        },
        {
          "given": "Jae-In",
          "family": "Hwang"
        }
      ],
      "sources": [
        "https://api.crossref.org/works/10.1145/3550082.3564175",
        "https://doi.org/10.1145/3550082.3564175",
        "assets/papers/flow-human.pdf"
      ],
      "doi": "10.1145/3550082.3564175",
      "publication_date": "2022-12-13",
      "page": "1-2",
      "tagline": "No-code conversation flows coordinate digital-human behavior.",
      "summary": "Flow Human lets service designers create a conversation flow using an authoring tool. The system turns that flow into verbal and nonverbal behavior, presents a digital human in a kiosk interaction, and collects user feedback.",
      "pipeline": [
        "Conversation flow",
        "Behavior coordination",
        "Digital-human interaction"
      ],
      "facts": {
        "Input": "An authored conversation flow",
        "Output": "Speech, facial animation, co-speech gesture, and feedback collection",
        "Method": "Flow-based authoring and coordinated digital-human behavior",
        "Data and scope": "Kiosk use case described in the poster",
        "Evaluation": "System and use-case demonstration in a two-page poster",
        "Limitations": "Behavior depends on the authored flow and available modules; this poster does not establish unrestricted dialogue competence."
      },
      "abstract": "In this poster, we present Flow Human, a no-code system that generates conversational behavior of digital humans from the text. Our users only need to build a conversation flow they want to talk to customers using the flow-based authoring tool we developed. Our system then automatically generates the verbal and non-verbal behavior of digital humans along the conversation flow, interacts with customers, and collects feedback. We believe that this work can serve the potential to be distributed to various services that have not been introduced because of the challenging task of controlling multiple factors in digital humans (e.g., conversation flow, co-speech gestures, and facial animation).",
      "abstract_source": "assets/papers/flow-human.pdf",
      "artifact_status": "Independent implementation of the paper’s core ideas, with setup instructions and data preparation documented in the repository README. The institute’s original source, datasets and trained models are not distributed.",
      "related": [],
      "pdf_delivery": "external publisher or preprint archive",
      "local_source_pdf": "assets/papers/flow-human.pdf",
      "public_navigation": true,
      "code": "https://github.com/ghazanPK/flow-human",
      "method_figure": {
        "image": "assets/research/flow-human/method.png",
        "alt": "Method diagram from Figure 2 of the flow-human paper",
        "caption": "Method diagram from the paper · Figure 2, PDF page 2.",
        "width": 600,
        "height": 493,
        "responsive": [
          {
            "image": "assets/research/flow-human/method-600.webp",
            "width": 600,
            "height": 493
          }
        ]
      }
    },
    {
      "year": 2021,
      "type": "journal article",
      "title": "Silhouettes from Real Objects Enable Realistic Interactions with a Virtual Human in Mobile Augmented Reality",
      "venue": "Applied Sciences",
      "slug": "silhouette-mobile-ar",
      "short_title": "Silhouette AR",
      "image": "assets/works/silhouette-ar.webp",
      "status": "published",
      "authors": [
        {
          "given": "Hanseob",
          "family": "Kim"
        },
        {
          "given": "Ghazanfar",
          "family": "Ali"
        },
        {
          "given": "Andréas",
          "family": "Pastor"
        },
        {
          "given": "Myungho",
          "family": "Lee"
        },
        {
          "given": "Gerard J.",
          "family": "Kim"
        },
        {
          "given": "Jae-In",
          "family": "Hwang"
        }
      ],
      "sources": [
        "https://api.crossref.org/works/10.3390/app11062763",
        "https://doi.org/10.3390/app11062763",
        "assets/papers/silhouette-mobile-ar.pdf"
      ],
      "doi": "10.3390/app11062763",
      "publication_date": "2021-03-19",
      "volume": "11",
      "issue": "6",
      "page": "2763",
      "tagline": "Real-object silhouettes give virtual humans spatial context in mobile AR.",
      "summary": "A lightweight segmentation pipeline constructs silhouette geometry from camera images of real objects. A virtual character can interact with this geometry and be occluded by it without a pre-modeled object proxy. A mobile animal-doll scenario tests changing views and object deformation.",
      "pipeline": [
        "Camera image",
        "Segmentation + silhouette",
        "Spatially aware AR interaction"
      ],
      "facts": {
        "Input": "Device-camera images of real objects",
        "Output": "Silhouette geometry for virtual-human occlusion and interaction",
        "Method": "Segmentation and dynamic silhouette geometry",
        "Data and scope": "Mobile AR animal-doll scenario",
        "Evaluation": "Paper reports a 2.4M-parameter model, 0.971 mIoU, and a 24-person pilot study",
        "Limitations": "Silhouette geometry approximates visible shape; it is not a complete reconstruction of hidden 3D object geometry."
      },
      "abstract": ": Realistic interactions with real objects (e.g., animals, toys, robots) in an augmented reality (AR) environment enhances the user experience. The common AR apps on the market achieve realistic interactions by superimposing pre-modeled virtual proxies on the real objects in the AR environment. This way user perceives the interaction with virtual proxies as interaction with real objects. However, catering to environment change, shape deformation, and view update is not a trivial task. Our proposed method uses the dynamic silhouette of a real object to enable realistic interactions. Our approach is practical, lightweight, and requires no additional hardware besides the device camera. For a case study, we designed a mobile AR application to interact with real animal dolls. Our scenario included a virtual human performing four types of realistic interactions. Results demonstrated our method’s stability that does not require pre-modeled virtual proxies in case of shape deformation and view update. We also conducted a pilot study using our approach and reported significant improvements in user perception of spatial awareness and presence for realistic interactions with a virtual human.",
      "abstract_source": "assets/papers/silhouette-mobile-ar.pdf",
      "artifact_status": "Independent implementation of the paper’s core ideas, with setup instructions and data preparation documented in the repository README. The institute’s original source, datasets and trained models are not distributed.",
      "related": [],
      "video": {
        "id": "75P9iV8M8e8",
        "label": "Paper presentation / demo",
        "source": "https://sites.google.com/view/mrlabkist/video-demo"
      },
      "pdf_delivery": "external publisher or preprint archive",
      "local_source_pdf": "assets/papers/silhouette-mobile-ar.pdf",
      "public_navigation": true,
      "code": "https://github.com/ghazanPK/silhouette-mobile-ar",
      "method_figure": {
        "image": "assets/research/silhouette-mobile-ar/graphical-abstract.png",
        "alt": "Graphical abstract: real-object segmentation and dynamic silhouette proxies enable virtual-human occlusion, walking and contact in mobile AR",
        "caption": "Graphical abstract diagram. Dynamic silhouette proxies support shape-aware occlusion, collision-aware walking and contact with real objects.",
        "width": 1774,
        "height": 887,
        "responsive": [
          {
            "image": "assets/research/silhouette-mobile-ar/graphical-abstract-640.webp",
            "width": 640,
            "height": 320
          },
          {
            "image": "assets/research/silhouette-mobile-ar/graphical-abstract-1280.webp",
            "width": 1280,
            "height": 640
          },
          {
            "image": "assets/research/silhouette-mobile-ar/graphical-abstract-1774.webp",
            "width": 1774,
            "height": 887
          }
        ]
      }
    },
    {
      "year": 2021,
      "type": "adjunct paper",
      "title": "ASAP: Auto-generating Storyboard And Previz with Virtual Humans",
      "venue": "IEEE ISMAR-Adjunct",
      "slug": "asap-ismar",
      "short_title": "ASAP ISMAR",
      "image": "assets/works/asap.webp",
      "status": "published",
      "authors": [
        {
          "given": "Hanseob",
          "family": "Kim"
        },
        {
          "given": "Ghazanfar",
          "family": "Ali"
        },
        {
          "given": "Jae-In",
          "family": "Hwang"
        }
      ],
      "sources": [
        "https://api.crossref.org/works/10.1109/ismar-adjunct54149.2021.00071",
        "https://doi.org/10.1109/ismar-adjunct54149.2021.00071",
        "https://www.researchgate.net/publication/355894656_ASAP_Auto-generating_Storyboard_And_Previz_with_Virtual_Humans"
      ],
      "doi": "10.1109/ismar-adjunct54149.2021.00071",
      "publication_date": "2021-10",
      "page": "316-320",
      "tagline": "An early ASAP tool turns script structure into animated previews.",
      "summary": "The ISMAR paper presents a screenplay-driven tool for screenwriters and filmmakers. It parses character, dialogue, and action paragraphs and combines behavior-generation methods to animate virtual humans; users can capture played scenes into a storyboard.",
      "pipeline": [
        "Final Draft screenplay",
        "Paragraph parsing + behavior",
        "Previz + storyboard"
      ],
      "facts": {
        "Input": "A screenplay in Final Draft format",
        "Output": "Previsualized virtual-human animation and captured storyboards",
        "Method": "Screenplay parsing with learned, data-driven, and rule-based behavior modules",
        "Data and scope": "System scenarios in the ISMAR paper",
        "Evaluation": "Tool and workflow demonstration",
        "Limitations": "The 2024-online journal is a later extension. Its multi-output benchmarks are not results of this 2021 paper."
      },
      "artifact_status": "Independent implementation of the paper’s core ideas, with setup instructions and data preparation documented in the repository README. The institute’s original source, datasets and trained models are not distributed.",
      "related": [
        "asap-journal",
        "asap-live"
      ],
      "related_video": {
        "id": "omdEg7Ro_bU",
        "label": "Related ASAP system: SIGGRAPH Asia 2022 live presentation",
        "source": "https://sites.google.com/view/mrlabkist/video-demo"
      },
      "abstract": "We present a tool for Auto-generating Storyboard And Previz for screenwriters and filmmakers, called ASAP. Our system allows users to easily simulate their stories in the form of 3D animated/visual scenes with virtual humans in a virtual environment. We only ask users to write their script using Final Draft, an exclusive screenwriting tool, and upload them to our system. The uploaded script is parsed into paragraphs of the action, character, and dialogue. From those paragraphs (i.e., text data), our system uses a combination of deep learning, data-driven, and rule-based approaches to instantly generate virtual human’s physical motions and co-speech gestures, presenting natural behavior/dialogue scenes. Thus, users can observe automatically generated pre-visualized animations (i.e., previz) from the script and can create the storyboard by capturing scenes being played. Our ASAP can minimize time-and-money consuming and labor-intensive work in the early stages of filmmaking, and do it as soon as possible. We believe that our tool and approach have a good potential for wide dissemination in the film industry.",
      "abstract_source": "https://www.researchgate.net/publication/355894656_ASAP_Auto-generating_Storyboard_And_Previz_with_Virtual_Humans",
      "pdf_delivery": "external publisher or preprint archive",
      "public_navigation": true,
      "code": "https://github.com/ghazanPK/asap-ismar",
      "method_figure": {
        "image": "assets/research/asap-ismar/method.svg",
        "alt": "Scientific method schematic for asap-ismar",
        "caption": "Graphical abstract diagram. Screenplay structure drives coordinated virtual-actor behavior and previsualization.",
        "width": 1500,
        "height": 610
      }
    },
    {
      "year": 2021,
      "type": "workshop paper",
      "title": "Auto-generating Virtual Human Behavior by Understanding User Contexts",
      "venue": "IEEE VR Abstracts and Workshops",
      "slug": "context-aware-behavior",
      "short_title": "Context-aware behavior",
      "image": "assets/works/context-behavior.webp",
      "status": "published",
      "authors": [
        {
          "given": "Hanseob",
          "family": "Kim"
        },
        {
          "given": "Ghazanfar",
          "family": "Ali"
        },
        {
          "given": "Seungwon",
          "family": "Kim"
        },
        {
          "given": "Gerard J.",
          "family": "Kim"
        },
        {
          "given": "Jae-In",
          "family": "Hwang"
        }
      ],
      "sources": [
        "https://api.crossref.org/works/10.1109/vrw52623.2021.00178",
        "https://doi.org/10.1109/vrw52623.2021.00178",
        "assets/papers/context-aware-behavior.pdf"
      ],
      "doi": "10.1109/vrw52623.2021.00178",
      "publication_date": "2021-03",
      "page": "591-592",
      "tagline": "Natural-language context activates grounded virtual-human actions.",
      "summary": "A BERT-based model jointly classifies whether a sentence requests conversation or action and extracts action entities. An interaction module turns those predictions into virtual-human behaviors in a controlled room scenario.",
      "pipeline": [
        "Natural-language input",
        "Sentence + entity classifier",
        "Grounded room behavior"
      ],
      "facts": {
        "Input": "Natural-language conversation and action requests",
        "Output": "Action intent, extracted entities, and virtual-human behavior",
        "Method": "Joint sentence classification and entity classification",
        "Data and scope": "Controlled room scenario with a fixed object and interaction set",
        "Evaluation": "Pilot study of perceived naturalness and user experience",
        "Limitations": "The controlled object and action inventory limits generalization to arbitrary environments."
      },
      "abstract": "Virtual humans are most natural and effective when it can act out and animate verbal/gestural actions. One popular method to realize this is to infer the actions from predefined phrases. This research aims to provide a more flexible method to activate various behaviors straight from natural conversations. Our approach uses BERT as the backbone for natural language understanding and, on top of it, a jointly learned sentence classifier (SC) and entity classifier (EC). The SC classifies the input into conversation or action, and EC extracts the entities for the action. The pilot study has shown promising results with high perceived naturalness and positive experiences.",
      "abstract_source": "assets/papers/context-aware-behavior.pdf",
      "artifact_status": "Independent implementation of the paper’s core ideas, with setup instructions and data preparation documented in the repository README. The institute’s original source, datasets and trained models are not distributed.",
      "related": [
        "wearable-mr-agent"
      ],
      "pdf_delivery": "external publisher or preprint archive",
      "local_source_pdf": "assets/papers/context-aware-behavior.pdf",
      "public_navigation": true,
      "code": "https://github.com/ghazanPK/context-aware-behavior",
      "method_figure": {
        "image": "assets/research/context-aware-behavior/method.png",
        "alt": "Method diagram from Figure 2 of the context-aware-behavior paper",
        "caption": "Method diagram from the paper · Figure 2, PDF page 2.",
        "width": 746,
        "height": 649,
        "responsive": [
          {
            "image": "assets/research/context-aware-behavior/method-640.webp",
            "width": 640,
            "height": 557
          },
          {
            "image": "assets/research/context-aware-behavior/method-746.webp",
            "width": 746,
            "height": 649
          }
        ]
      }
    },
    {
      "year": 2020,
      "type": "journal article",
      "title": "Automatic text‐to‐gesture rule generation for embodied conversational agents",
      "venue": "Computer Animation and Virtual Worlds",
      "role": "first author",
      "slug": "automatic-text-to-gesture",
      "short_title": "Automatic gesture rules",
      "image": "assets/works/automatic-rules.webp",
      "status": "published",
      "authors": [
        {
          "given": "Ghazanfar",
          "family": "Ali"
        },
        {
          "given": "Myungho",
          "family": "Lee"
        },
        {
          "given": "Jae‐In",
          "family": "Hwang"
        }
      ],
      "sources": [
        "https://api.crossref.org/works/10.1002/cav.1944",
        "https://doi.org/10.1002/cav.1944",
        "assets/papers/automatic-text-to-gesture.pdf"
      ],
      "doi": "10.1002/cav.1944",
      "publication_date": "2020-09-06",
      "volume": "31",
      "issue": "4-5",
      "article_number": "e1944",
      "tagline": "Automatically mined rules reduce manual co-speech gesture authoring.",
      "summary": "The method mines text-to-gesture mappings from public video rather than requiring experts to author every rule. At runtime, word embeddings search for semantically relevant rules and activate recorded gesture units. User evaluation compares mined maps with manual maps and their combination.",
      "pipeline": [
        "Public video + pose",
        "Automatic rule mining",
        "Semantic gesture retrieval"
      ],
      "facts": {
        "Input": "Offline video and pose; runtime text",
        "Output": "Retrieved co-speech gestures",
        "Method": "Automated video-to-rule mapping with semantic word-embedding search",
        "Data and scope": "Approximately 106 hours of public video",
        "Evaluation": "Comparison with manual rule maps; gesture variety and user perception",
        "Limitations": "Rule retrieval depends on corpus coverage and the gesture inventory; automatic mapping does not directly decode novel motion."
      },
      "abstract": "Interactions with embodied conversational agents (ECAs) can be enhanced using humanlike co-speech gestures. Traditionally rulebased co-speech gesture mapping has been utilized for this purpose. However, the creation of this mapping is laborious and often requires human experts. Moreover, human-created mapping tends to be limited, therefore prone to generate repeated gestures. In this paper, we present an approach to automate the generation of rule-based co-speech gesture mapping from publicly available large video dataset without the intervention of human experts. At runtime, word embedding is utilized for rule searching to get the semantic-aware, meaningful, and accurate rule. The evaluation indicated that our method achieved comparable performance with the manual map generated by human experts, with a more variety of gestures activated. Moreover, synergy effects were observed in users’ perception of generated co-speech gestures when combined with the manual map.",
      "abstract_source": "assets/papers/automatic-text-to-gesture.pdf",
      "artifact_status": "Independent implementation of the paper’s core ideas, with setup instructions and data preparation documented in the repository README. The institute’s original source, datasets and trained models are not distributed.",
      "related": [
        "multilingual-gesture",
        "ridge",
        "wild-pose-matching"
      ],
      "video": {
        "id": "GIxaI9yTmMc",
        "label": "Paper presentation / demo",
        "source": "https://sites.google.com/view/mrlabkist/video-demo"
      },
      "pdf_delivery": "external publisher or preprint archive",
      "local_source_pdf": "assets/papers/automatic-text-to-gesture.pdf",
      "public_navigation": true,
      "code": "https://github.com/ghazanPK/automatic-text-to-gesture",
      "method_figure": {
        "image": "assets/research/automatic-text-to-gesture/graphical-abstract.png",
        "alt": "Graphical abstract: mining text–gesture rules from video and retrieving recorded motion for new text",
        "caption": "Graphical abstract diagram. Video-derived rules map new text to recorded co-speech gestures.",
        "width": 1774,
        "height": 887,
        "responsive": [
          {
            "image": "assets/research/automatic-text-to-gesture/graphical-abstract-640.webp",
            "width": 640,
            "height": 320
          },
          {
            "image": "assets/research/automatic-text-to-gesture/graphical-abstract-1280.webp",
            "width": 1280,
            "height": 640
          },
          {
            "image": "assets/research/automatic-text-to-gesture/graphical-abstract-1774.webp",
            "width": 1774,
            "height": 887
          }
        ]
      }
    },
    {
      "year": 2019,
      "type": "conference paper",
      "title": "Design of Seamless Multi-modal Interaction Framework for Intelligent Virtual Agents in Wearable Mixed Reality Environment",
      "venue": "CASA",
      "role": "first author",
      "slug": "wearable-mr-agent",
      "short_title": "Wearable MR agent",
      "image": "assets/works/wearable-mr.webp",
      "status": "published",
      "authors": [
        {
          "given": "Ghazanfar",
          "family": "Ali"
        },
        {
          "given": "Hong-Quan",
          "family": "Le"
        },
        {
          "given": "Junho",
          "family": "Kim"
        },
        {
          "given": "Seung-Won",
          "family": "Hwang"
        },
        {
          "given": "Jae-In",
          "family": "Hwang"
        }
      ],
      "sources": [
        "https://api.crossref.org/works/10.1145/3328756.3328758",
        "https://doi.org/10.1145/3328756.3328758",
        "assets/papers/wearable-mr-agent.pdf"
      ],
      "doi": "10.1145/3328756.3328758",
      "publication_date": "2019-07",
      "page": "47-52",
      "tagline": "A modular virtual guide combines speech, gaze, and spatial context in wearable MR.",
      "summary": "The framework integrates spatial mapping, speech recognition, gaze, object recognition, a domain-specific chatbot, and virtual-character animation. Computationally intensive components run on a cloud platform to support a wearable device with limited resources.",
      "pipeline": [
        "Speech + gaze + scene",
        "Modular agent / cloud",
        "Embodied MR response"
      ],
      "facts": {
        "Input": "Speech, gaze, recognized objects, and spatial mapping",
        "Output": "An interactive virtual guide with verbal and animated responses",
        "Method": "Modular interaction framework with cloud-assisted processing",
        "Data and scope": "Wearable mixed-reality application scenarios",
        "Evaluation": "Paper reports responses within 2–4 seconds in its tests",
        "Limitations": "The response measurements describe the tested hardware, network, and scenarios; cloud connectivity and spatial mapping are required by the design."
      },
      "abstract": "In this paper, we present the design of a multimodal interaction framework for intelligent virtual agents in wearable mixed reality environments, especially for interactive applications at museums, botanical gardens, and similar places. These places need engaging and no-repetitive digital content delivery to maximize user involvement. An intelligent virtual agent is a promising mode for both purposes. Premises of framework is wearable mixed reality provided by MR devices supporting spatial mapping. We envisioned a seamless interaction framework by integrating potential features of spatial mapping, virtual character animations, speech recognition, gazing, domain-specific chatbot and object recognition to enhance virtual experiences and communication between users and virtual agents. By applying a modular approach and deploying computationally intensive modules on cloud-platform, we achieved a seamless virtual experience in a device with limited resources. Human-like gaze and speech interaction with a virtual agent made it more interactive. Automated mapping of body animations with the content of a speech made it more engaging. In our tests, the virtual agents responded within 2-4 seconds after the user query. The strength of the framework is flexibility and adaptability. It can be adapted to any wearable MR device supporting spatial mapping.",
      "abstract_source": "assets/papers/wearable-mr-agent.pdf",
      "artifact_status": "Independent implementation of the paper’s core ideas, with setup instructions and data preparation documented in the repository README. The institute’s original source, datasets and trained models are not distributed.",
      "related": [
        "context-aware-behavior"
      ],
      "preprint": "https://arxiv.org/abs/2503.19334",
      "video": {
        "id": "zlmVpUgBdew",
        "label": "Paper presentation / demo",
        "source": "https://sites.google.com/view/mrlabkist/video-demo"
      },
      "pdf_delivery": "external publisher or preprint archive",
      "local_source_pdf": "assets/papers/wearable-mr-agent.pdf",
      "public_navigation": true,
      "code": "https://github.com/ghazanPK/wearable-mr-agent",
      "method_figure": {
        "image": "assets/research/wearable-mr-agent/wearable-system.png",
        "alt": "Wearable MR Agent architecture: camera, gaze and speech inputs, object-grounded dialogue, coordinated agent behavior, spatial anchors and engagement loop",
        "caption": "Graphical abstract diagram. Camera, gaze and speech inputs support object-grounded conversation and coordinated virtual-agent behavior.",
        "width": 1774,
        "height": 887,
        "responsive": [
          {
            "image": "assets/research/wearable-mr-agent/wearable-system-640.webp",
            "width": 640,
            "height": 320
          },
          {
            "image": "assets/research/wearable-mr-agent/wearable-system-1280.webp",
            "width": 1280,
            "height": 640
          },
          {
            "image": "assets/research/wearable-mr-agent/wearable-system-1774.webp",
            "width": 1774,
            "height": 887
          }
        ]
      }
    },
    {
      "slug": "doctoral-thesis",
      "short_title": "Doctoral thesis",
      "title": "Scalable Hybrid Approach of Co-speech Text-to-Gesture Generation for Interactive Digital Humans",
      "authors": [
        {
          "given": "Ghazanfar",
          "family": "Ali"
        }
      ],
      "year": 2023,
      "publication_date": "2023-08",
      "type": "doctoral thesis",
      "status": "doctoral thesis",
      "venue": "University of Science and Technology, KIST School",
      "image": "assets/works/automatic-rules.webp",
      "tagline": "The evolution of text-driven gesture systems for interactive digital humans.",
      "summary": "The dissertation connects wearable mixed-reality agents, automatic rule mining, GestureCLR pose matching, multilingual gesture retrieval, and ConGRets speaker-style adaptation. It documents the research lineage and should be cited as a dissertation rather than as a journal article.",
      "abstract": "This thesis chronicles developing and optimizing co-speech text-to-gesture generation for digital humans, underlining a significant leap towards enhancing user engagement and interaction in wearable mixed reality environments. The research narrative begins with the conception of a multimodal interaction framework for intelligent virtual agents and concludes with the advanced co-speech gesture generation system, ConGRets, capable of zero-shot speaker style adaptation. The research began with creating an intelligent virtual agent system for mixedreality environments. This system integrated features such as speech recognition, domainspecific chatbot capabilities, and manual mapping of body animations to speech content, highlighting the potential of digital humans as interactive tools. However, it also emphasized the need for a more automated and expressive body gesture generation system to enhance communication and engagement further. Motivated by this requirement, the research introduced an automated, rule-based co-speech gesture mapping system. This system, leveraging large publicly available video datasets and word embeddings, generated a diverse and accurate set of gestures that bypassed the limitations of traditional human-created mappings. A significant advancement was marked by the introduction of GestureCLR, a contrastive learning model that drastically improved the matching of 2D poses from videos to 3D gesture units, outperforming the previous mean cosine similarity method. The integration of GestureCLR and K-Means clustering on gesture units derived from English monologue videos and Korean speaker motion capture sequences led to the development of a cross-lingual method. This method effectively generated realistic, humanlike gestures for digital humans, augmenting their non-verbal communication capabiliii ities across different languages without compromising user experience. The final phase of the research introduced ConGRets, a unique framework optimizing real-time performance and resource usage. Addressing the limitations of previous methods, such as slow query speeds and averaged gesture generation, ConGRets incorporated zero-shot speaker style adaptation. This novel feature allowed the system to generate gestures in alignment with the text input and the individual speaker ’s motion style. The thesis narrates the journey of this groundbreaking research, which is now permeating diverse applications, including medical screening kiosks, the virtual therapy system, the actor-focused pre-visualization system ASAP , the ChatGPT frontend, and the virtual presenter. This comprehensive exploration of the evolution of co-speech text-to-gesture generation for digital humans underscores a significant stride towards enriching their non-verbal communication abilities and enhancing user interaction and engagement in various environments.",
      "abstract_source": "assets/papers/ghazanfar-ali-doctoral-thesis.pdf",
      "facts": {
        "Input": "Text, video-derived pose, and captured gesture data across the included systems",
        "Output": "Text-driven co-speech gestures for interactive digital humans",
        "Method": "Multimodal agents, automated rule maps, contrastive matching, and style-aware retrieval",
        "Data and scope": "Research conducted for the 2023 AI-Robotics dissertation",
        "Evaluation": "System evaluations and user studies across dissertation chapters",
        "Limitations": "Chapter results reflect their respective systems and datasets; a dissertation chapter is not evidence of a later journal publication."
      },
      "pipeline": [
        "Interactive MR agent",
        "Automatic maps + GestureCLR",
        "Style-aware gesture retrieval"
      ],
      "related": [
        "wearable-mr-agent",
        "automatic-text-to-gesture",
        "wild-pose-matching",
        "multilingual-gesture"
      ],
      "artifact_status": "Original project implementations are held by the institute. Separate educational repositories will be handled paper by paper.",
      "sources": [
        "assets/papers/ghazanfar-ali-doctoral-thesis.pdf"
      ],
      "pdf_delivery": "external publisher or preprint archive",
      "local_source_pdf": "assets/papers/ghazanfar-ali-doctoral-thesis.pdf",
      "availability_note": "An external manuscript archive link will be added when the public deposit is available.",
      "public_navigation": true
    }
  ]
}
