Towards human-centered and efficient video synthesis: a survey of multimodal diffusion models
{"Abstract":[0],"Multimodal":[1],"video":[2,12,222],"diffusion":[3],"models":[4],"have":[5,43],"emerged":[6],"as":[7,114,179],"transformative":[8],"tools":[9],"for":[10,135,176,219],"controlled":[11],"synthesis,":[13],"integrating":[14],"text,":[15],"images,":[16],"audio,":[17],"and":[18,37,56,77,94,106,161,196,214],"pose":[19],"sequences":[20],"to":[21,120,205],"generate":[22],"semantically":[23],"meaningful":[24],"content.":[25],"Despite":[26],"significant":[27],"advances,":[28],"critical":[29],"gaps":[30],"persist":[31],"in":[32,59,190],"temporal":[33,115],"consistency,":[34],"multimodal":[35,78,147,221],"alignment,":[36],"human-centric":[38,88,149],"motion":[39,61,89,166],"generation.":[40,223],"Existing":[41],"surveys":[42],"not":[44],"addressed":[45],"clearly":[46],"the":[47,83,186],"complex":[48],"interplay":[49],"between":[50,103],"these":[51],"components,":[52],"particularly":[53],"physiological":[54,92],"constraints":[55,157],"identity":[57,95,162],"preservation":[58,163],"human":[60],"synthesis.":[62],"This":[63],"survey":[64],"provides":[65],"a":[66,70,180],"comprehensive":[67],"analysis":[68,99,192],"through":[69],"unified":[71],"architectural":[72],"framework,":[73],"examining":[74],"spatial-temporal":[75],"representations":[76],"conditioning":[79],"mechanisms.":[80],"We":[81,168],"present":[82],"first":[84],"systematic":[85],"evaluation":[86,212],"of":[87,201],"modeling,":[90],"addressing":[91],"plausibility":[93],"consistency":[96],"challenges.":[97],"Our":[98],"reveals":[100],"fundamental":[101],"trade-offs":[102],"computational":[104,123],"efficiency":[105],"generation":[107],"quality,":[108],"with":[109,128,145,165,173],"reported":[110],"specialized":[111],"techniques":[112],"such":[113],"block":[116],"pruning":[117],"achieving":[118],"up":[119],"$$523\\\\times":[121],"$$":[122],"savings":[124],"under":[125],"specific":[126],"baselines":[127],"minimal":[129],"quality":[130],"degradation":[131],"(see":[132],"Sect.":[133],"6.3":[134],"comparability":[136],"caveats).":[137],"Key":[138],"findings":[139],"indicate":[140],"that":[141,184],"current":[142],"approaches":[143],"struggle":[144],"seamless":[146],"integration,":[148],"applications":[150],"face":[151],"\\"uncanny":[152],"valley\\"":[153],"effects":[154],"when":[155],"physics":[156],"are":[158],"too":[159],"rigid,":[160],"conflicts":[164],"dynamics.":[167],"introduce":[169],"MIME-Vid":[170,202],"(Multi-modal":[171],"Integration":[172],"Motion":[174],"Enhancement":[175],"Video":[177],"Generation)":[178],"conceptual":[181],"reference":[182],"framework":[183],"operationalises":[185],"unifying":[187],"principles":[188],"identified":[189],"our":[191],"(reference-flexibility,":[193],"physics-perception":[194],"asymmetry,":[195],"hierarchical":[197],"disentanglement);":[198],"empirical":[199],"validation":[200],"is":[203],"deferred":[204],"follow-up":[206],"work.":[207],"Furthermore,":[208],"we":[209],"propose":[210],"novel":[211],"paradigms":[213],"identify":[215],"future":[216],"research":[217],"directions":[218],"advancing":[220]}
Authors
- Alaa Abdullah Albaghdadi
- Ahmad R. Naghsh-Nilchi
Institutions
- University of Isfahan (IR)
Publication Details
- Journal
- Artificial Intelligence Review
- Published
- 2026-09-13
- DOI
- https://doi.org/10.1007/s10462-026-11699-z
- Primary Topic
- Generative Adversarial Networks and Image Synthesis
- Type
- article
- Field-Weighted Citation Impact
- 0.00