[
  {
    "id": "10.3390/systems14101198",
    "type": "article-journal",
    "title": "Talking Face Generation in Socio-Technical Systems: A Systematic Review of Deep Learning Architectures, Deployment Contexts, and Governance Frameworks",
    "author": [
      {
        "family": "Salahudeen",
        "given": "Ridwan"
      },
      {
        "family": "Fonkam",
        "given": "Mathias"
      },
      {
        "family": "Vajjhala",
        "given": "Narasimha Rao"
      },
      {
        "family": "Junaidu",
        "given": "Sahalu Balarabe"
      },
      {
        "family": "Garba",
        "given": "Aliyu"
      }
    ],
    "container-title": "Systems",
    "issued": {
      "date-parts": [
        [
          2026,
          9,
          24
        ]
      ]
    },
    "volume": "14",
    "issue": "10",
    "number": "1198",
    "publisher": "MDPI",
    "ISSN": "2079-8954",
    "DOI": "10.3390/systems14101198",
    "URL": "https://www.narasimharao.net/research/talking-face-generation-socio-technical-systems-systematic-review/",
    "abstract": "Talking face generation (TFG)—the synthesis of photorealistic speaking video from a portrait and an audio signal—has passed through five architectural generations since 2016, from Long Short-Term Memory (LSTM) lip-sync models through Generative Adversarial Networks (GANs), Neural Radiance Fields (NeRFs) and Diffusion Models to Diffusion Transformers (DiT) and Gaussian Splatting, and now enters healthcare, education, identity management and public communication in real time. This systematic review synthesises 92 primary studies (January 2016–December 2025), selected from 18,343 records across six repositories under Kitchenham’s guidelines and PRISMA 2020, addressing architecture evolution, dataset bias, evaluation metrics, paradigm shifts, ethical safeguards and socio-technical deployment. DiT and Gaussian Splatting account for roughly 36% of studies dated 2024 or later, and GAN use has fallen below 3%; training corpora are demographically skewed and under-documented, and the dominant metrics (PSNR, SSIM, SyncNet) do not measure what deployment requires. Only one of the 92 systems embeds any technical safeguard, a non-compliant post hoc watermark, and none implements consent or provenance, although the EU AI Act, the TAKE IT DOWN Act, China’s labelling Measures and C2PA v2.0 were all adopted within the review period. We propose three deployment-fitness metrics and the RTFG-STS Framework as research constructs requiring empirical validation.",
    "keyword": "talking face generation, socio-technical systems, deepfake governance, systematic literature review, responsible AI, human-AI interaction, talking head generation, deepfakes, generative AI, deep learning, computer vision, diffusion models, Diffusion Transformers, Gaussian splatting, neural radiance fields, dataset bias, evaluation metrics, AI governance, EU AI Act, content provenance (C2PA), identity management, systematic review",
    "language": "en"
  }
]