Image-Text-to-Video
MiniMax H3
Diffusion Single File
English
Chinese
ref2va
comfyui
int8-convrot
synchronized-audio-video
experimental
Instructions to use Wesley1234/minimax_h3_ref2va_patchin_hf102 with libraries, inference providers, notebooks, and local apps. Follow these links to get started.
- Libraries
- Diffusion Single File
How to use Wesley1234/minimax_h3_ref2va_patchin_hf102 with Diffusion Single File:
# No code snippets available yet for this library. # To use this model, check the repository files and the library's documentation. # Want to help? PRs adding snippets are welcome at: # https://github.com/huggingface/huggingface.js
- Notebooks
- Google Colab
- Kaggle
| { | |
| "controls": { | |
| "seed": 2608103502, | |
| "steps": 20, | |
| "size": [ | |
| 512, | |
| 512 | |
| ], | |
| "frames": 22, | |
| "prompt": "centered front-facing portrait; identical for all candidates", | |
| "baseline": "minimax_h3_ref2va_int8_convrot.safetensors" | |
| }, | |
| "baseline_regions": { | |
| "forehead": { | |
| "gradient": 19.118993974634726, | |
| "laplacian": 16.466526990591827, | |
| "highpass": 0.8236231262033636 | |
| }, | |
| "left_cheek": { | |
| "gradient": 20.42510201739831, | |
| "laplacian": 26.98414471590909, | |
| "highpass": 1.0676020979881287 | |
| }, | |
| "right_cheek": { | |
| "gradient": 34.39876539933298, | |
| "laplacian": 207.0204395568182, | |
| "highpass": 2.35250105641105 | |
| }, | |
| "portrait": { | |
| "gradient": 29.842277256068837, | |
| "laplacian": 86.62322641690345, | |
| "highpass": 1.685757739977403 | |
| } | |
| }, | |
| "results": { | |
| "patchin_hf102": { | |
| "label": "Patch input 1.02", | |
| "frame_shape": [ | |
| 22, | |
| 512, | |
| 512, | |
| 3 | |
| ], | |
| "rgb_mae_255": 14.241889953613281, | |
| "psnr_db": 20.817959207130038, | |
| "parity_abs_diff": [ | |
| 14.241340637207031, | |
| 14.256017684936523, | |
| 14.226961135864258, | |
| 14.243230819702148 | |
| ], | |
| "parity_relative_spread": 0.00204021755758276, | |
| "region_ratios": { | |
| "forehead": { | |
| "gradient": 1.032062507677407, | |
| "laplacian": 1.1970488793243081, | |
| "highpass": 1.038160105570075 | |
| }, | |
| "left_cheek": { | |
| "gradient": 1.0283841064243358, | |
| "laplacian": 0.9345722788494126, | |
| "highpass": 0.9861274339583423 | |
| }, | |
| "right_cheek": { | |
| "gradient": 1.036924744809861, | |
| "laplacian": 1.0964563953998219, | |
| "highpass": 1.0323794535268038 | |
| }, | |
| "portrait": { | |
| "gradient": 1.1443190568533455, | |
| "laplacian": 1.2320932420919597, | |
| "highpass": 1.193404968594918 | |
| } | |
| }, | |
| "skin_mean_ratios": { | |
| "gradient": 1.0324571196372012, | |
| "laplacian": 1.0760258511911809, | |
| "highpass": 1.018888997685074 | |
| }, | |
| "skin_highpass_positive_regions": 2, | |
| "temporal_ratio": 1.0321481634935248, | |
| "audio_correlation": 0.9930406616019274, | |
| "audio_rms_ratio": 0.9189266010030337, | |
| "audio_samples_compared": 26624, | |
| "audio_rate": 32000 | |
| }, | |
| "patchio_hf101": { | |
| "label": "Patch I/O 1.01", | |
| "frame_shape": [ | |
| 22, | |
| 512, | |
| 512, | |
| 3 | |
| ], | |
| "rgb_mae_255": 26.415775299072266, | |
| "psnr_db": 15.82615784912672, | |
| "parity_abs_diff": [ | |
| 26.458765029907227, | |
| 26.356605529785156, | |
| 26.47501564025879, | |
| 26.372785568237305 | |
| ], | |
| "parity_relative_spread": 0.004482549917521291, | |
| "region_ratios": { | |
| "forehead": { | |
| "gradient": 0.9444839764533365, | |
| "laplacian": 0.8272422765181436, | |
| "highpass": 0.9609261260659551 | |
| }, | |
| "left_cheek": { | |
| "gradient": 0.9498798623889819, | |
| "laplacian": 0.7533049208713888, | |
| "highpass": 0.9362137479998064 | |
| }, | |
| "right_cheek": { | |
| "gradient": 0.8615758885295116, | |
| "laplacian": 0.6397598914467637, | |
| "highpass": 0.7811897209359578 | |
| }, | |
| "portrait": { | |
| "gradient": 0.9699411210561822, | |
| "laplacian": 0.7818911547345777, | |
| "highpass": 0.9346941424916256 | |
| } | |
| }, | |
| "skin_mean_ratios": { | |
| "gradient": 0.91864657579061, | |
| "laplacian": 0.7401023629454321, | |
| "highpass": 0.8927765316672397 | |
| }, | |
| "skin_highpass_positive_regions": 0, | |
| "temporal_ratio": 0.7695742329812562, | |
| "audio_correlation": 0.6777213227155774, | |
| "audio_rms_ratio": 0.9347905590619777, | |
| "audio_samples_compared": 26624, | |
| "audio_rate": 32000 | |
| }, | |
| "attn_last8_g102_backupAB": { | |
| "label": "Attention 1.02", | |
| "frame_shape": [ | |
| 22, | |
| 512, | |
| 512, | |
| 3 | |
| ], | |
| "rgb_mae_255": 24.50489044189453, | |
| "psnr_db": 16.033918404344604, | |
| "parity_abs_diff": [ | |
| 24.49074935913086, | |
| 24.49752426147461, | |
| 24.511693954467773, | |
| 24.51958656311035 | |
| ], | |
| "parity_relative_spread": 0.0011767939257850033, | |
| "region_ratios": { | |
| "forehead": { | |
| "gradient": 0.9651551361715502, | |
| "laplacian": 0.8632519050145561, | |
| "highpass": 0.9715350187455672 | |
| }, | |
| "left_cheek": { | |
| "gradient": 0.895689344936705, | |
| "laplacian": 0.6889516854452177, | |
| "highpass": 0.8737009644258836 | |
| }, | |
| "right_cheek": { | |
| "gradient": 0.9698874011966927, | |
| "laplacian": 0.8081264950412943, | |
| "highpass": 0.8652700024587207 | |
| }, | |
| "portrait": { | |
| "gradient": 0.9997757169194811, | |
| "laplacian": 1.0655103462302664, | |
| "highpass": 0.9821616419372078 | |
| } | |
| }, | |
| "skin_mean_ratios": { | |
| "gradient": 0.9435772941016493, | |
| "laplacian": 0.7867766951670226, | |
| "highpass": 0.9035019952100573 | |
| }, | |
| "skin_highpass_positive_regions": 0, | |
| "temporal_ratio": 0.9440205176611185, | |
| "audio_correlation": 0.6593426872963273, | |
| "audio_rms_ratio": 0.976732972125702, | |
| "audio_samples_compared": 26624, | |
| "audio_rate": 32000 | |
| }, | |
| "attn_last8_g105_backupAB": { | |
| "label": "Attention 1.05", | |
| "frame_shape": [ | |
| 22, | |
| 512, | |
| 512, | |
| 3 | |
| ], | |
| "rgb_mae_255": 26.17381477355957, | |
| "psnr_db": 15.466375766270218, | |
| "parity_abs_diff": [ | |
| 26.163225173950195, | |
| 26.164703369140625, | |
| 26.182741165161133, | |
| 26.184694290161133 | |
| ], | |
| "parity_relative_spread": 0.000820250883745451, | |
| "region_ratios": { | |
| "forehead": { | |
| "gradient": 0.9959880421304674, | |
| "laplacian": 0.8540994665054672, | |
| "highpass": 0.988992733340178 | |
| }, | |
| "left_cheek": { | |
| "gradient": 0.9608255294610666, | |
| "laplacian": 0.8020721975955389, | |
| "highpass": 0.9816722853274565 | |
| }, | |
| "right_cheek": { | |
| "gradient": 0.9779253911003554, | |
| "laplacian": 0.7476207497429617, | |
| "highpass": 0.8529531788916929 | |
| }, | |
| "portrait": { | |
| "gradient": 1.004480955684384, | |
| "laplacian": 0.9810051504775767, | |
| "highpass": 0.9774153908587273 | |
| } | |
| }, | |
| "skin_mean_ratios": { | |
| "gradient": 0.9782463208972964, | |
| "laplacian": 0.8012641379479893, | |
| "highpass": 0.9412060658531091 | |
| }, | |
| "skin_highpass_positive_regions": 0, | |
| "temporal_ratio": 0.7998909601840626, | |
| "audio_correlation": 0.7073727262310824, | |
| "audio_rms_ratio": 1.0813777891956717, | |
| "audio_samples_compared": 26624, | |
| "audio_rate": 32000 | |
| }, | |
| "adaln_video_g099": { | |
| "label": "Video AdaLN 0.99", | |
| "frame_shape": [ | |
| 22, | |
| 512, | |
| 512, | |
| 3 | |
| ], | |
| "rgb_mae_255": 9.915425300598145, | |
| "psnr_db": 23.4676943237718, | |
| "parity_abs_diff": [ | |
| 9.92910385131836, | |
| 9.926578521728516, | |
| 9.905007362365723, | |
| 9.90101146697998 | |
| ], | |
| "parity_relative_spread": 0.0028332001388467164, | |
| "region_ratios": { | |
| "forehead": { | |
| "gradient": 0.9991286249222505, | |
| "laplacian": 1.0026000155229857, | |
| "highpass": 1.006577991424609 | |
| }, | |
| "left_cheek": { | |
| "gradient": 0.9780371131275568, | |
| "laplacian": 0.9120942549436786, | |
| "highpass": 0.9648077761342945 | |
| }, | |
| "right_cheek": { | |
| "gradient": 1.0072948081898199, | |
| "laplacian": 0.978957933377793, | |
| "highpass": 1.0015376344560063 | |
| }, | |
| "portrait": { | |
| "gradient": 0.9180274832210044, | |
| "laplacian": 0.8724958236770269, | |
| "highpass": 0.8997279055195478 | |
| } | |
| }, | |
| "skin_mean_ratios": { | |
| "gradient": 0.9948201820798758, | |
| "laplacian": 0.9645507346148191, | |
| "highpass": 0.9909744673383032 | |
| }, | |
| "skin_highpass_positive_regions": 2, | |
| "temporal_ratio": 0.9975294373750192, | |
| "audio_correlation": 0.8806014978763488, | |
| "audio_rms_ratio": 0.9717682054851432, | |
| "audio_samples_compared": 26624, | |
| "audio_rate": 32000 | |
| }, | |
| "adaln_video_g101": { | |
| "label": "Video AdaLN 1.01", | |
| "frame_shape": [ | |
| 22, | |
| 512, | |
| 512, | |
| 3 | |
| ], | |
| "rgb_mae_255": 21.380828857421875, | |
| "psnr_db": 16.910562972665033, | |
| "parity_abs_diff": [ | |
| 21.360219955444336, | |
| 21.358842849731445, | |
| 21.403270721435547, | |
| 21.400985717773438 | |
| ], | |
| "parity_relative_spread": 0.002077930187772434, | |
| "region_ratios": { | |
| "forehead": { | |
| "gradient": 1.0077906715102303, | |
| "laplacian": 0.9818674456823687, | |
| "highpass": 1.0348544321311723 | |
| }, | |
| "left_cheek": { | |
| "gradient": 0.9018172697800978, | |
| "laplacian": 0.6945614191520033, | |
| "highpass": 0.8807424179496622 | |
| }, | |
| "right_cheek": { | |
| "gradient": 0.9685711252044322, | |
| "laplacian": 0.8277841438265235, | |
| "highpass": 0.8748947150154358 | |
| }, | |
| "portrait": { | |
| "gradient": 1.0097364352355103, | |
| "laplacian": 1.0435720605395844, | |
| "highpass": 0.9943788850381092 | |
| } | |
| }, | |
| "skin_mean_ratios": { | |
| "gradient": 0.9593930221649201, | |
| "laplacian": 0.8347376695536318, | |
| "highpass": 0.9301638550320902 | |
| }, | |
| "skin_highpass_positive_regions": 1, | |
| "temporal_ratio": 1.0230604475597387, | |
| "audio_correlation": 0.8226417228875448, | |
| "audio_rms_ratio": 0.9860379990897791, | |
| "audio_samples_compared": 26624, | |
| "audio_rate": 32000 | |
| } | |
| } | |
| } |