{"schema_version":1,"cases":[{"id":"real_identity","title":"원본 동일 입력","description":"Exact source copy: extraction determinism check, not an independent test.","group":"real_dance","referenceId":"real_floss","candidateId":"real_identity","expected":"aligned","expectation_source":"Deterministic transformation stress expectation, not independent correctness ground truth.","status":"aligned","flags":[],"syncMean":0.0,"dtwMean":0.0,"coverage":0.94595,"outcome":"expected_diagnostic","comparisonWallSeconds":0.03047,"maxProgressDeviation":0.0,"longestRepeatedReferenceSeconds":0.0,"dataUrl":"data/real_identity.json"},{"id":"real_slow_135","title":"전체 속도 1.35배 느리게","description":"Whole sequence slowed to1.35x duration with repeated original frames.","group":"real_dance","referenceId":"real_floss","candidateId":"real_slow_135","expected":"aligned","expectation_source":"Deterministic transformation stress expectation, not independent correctness ground truth.","status":"aligned","flags":[],"syncMean":0.05323,"dtwMean":0.02077,"coverage":0.90196,"outcome":"expected_diagnostic","comparisonWallSeconds":0.02876,"maxProgressDeviation":0.0332,"longestRepeatedReferenceSeconds":0.16667,"dataUrl":"data/real_slow_135.json"},{"id":"real_tempo_local","title":"구간별 속도 변화","description":"Nonuniform monotonic retiming: candidate45% corresponds to source22%.","group":"real_dance","referenceId":"real_floss","candidateId":"real_tempo_local","expected":"aligned","expectation_source":"Deterministic transformation stress expectation, not independent correctness ground truth.","status":"aligned","flags":[],"syncMean":0.1701,"dtwMean":0.04078,"coverage":0.88636,"outcome":"expected_diagnostic","comparisonWallSeconds":0.02317,"maxProgressDeviation":0.19444,"longestRepeatedReferenceSeconds":0.25,"dataUrl":"data/real_tempo_local.json"},{"id":"real_order_swap","title":"동작 구간 순서 교환","description":"Swap the two middle temporal quarters; endpoints preserved.","group":"real_dance","referenceId":"real_floss","candidateId":"real_order_swap","expected":"needs_review","expectation_source":"Deterministic transformation stress expectation, not independent correctness ground truth.","status":"aligned","flags":[],"syncMean":0.16094,"dtwMean":0.11355,"coverage":0.87805,"outcome":"unexpected_diagnostic","comparisonWallSeconds":0.02289,"maxProgressDeviation":0.11111,"longestRepeatedReferenceSeconds":0.16667,"dataUrl":"data/real_order_swap.json"},{"id":"real_skip_middle","title":"중간 동작 생략","description":"Remove source40–65%; no interpolation across the resulting cut.","group":"real_dance","referenceId":"real_floss","candidateId":"real_skip_middle","expected":"needs_review","expectation_source":"Deterministic transformation stress expectation, not independent correctness ground truth.","status":"aligned","flags":[],"syncMean":0.18604,"dtwMean":0.07074,"coverage":0.85714,"outcome":"unexpected_diagnostic","comparisonWallSeconds":0.01851,"maxProgressDeviation":0.16162,"longestRepeatedReferenceSeconds":0.08333,"dataUrl":"data/real_skip_middle.json"},{"id":"real_freeze_middle","title":"중간 프레임 정지","description":"Replace source30–65% with the frame at30%; jump to65% afterward.","group":"real_dance","referenceId":"real_floss","candidateId":"real_freeze_middle","expected":"needs_review","expectation_source":"Deterministic transformation stress expectation, not independent correctness ground truth.","status":"aligned","flags":[],"syncMean":0.08354,"dtwMean":0.08354,"coverage":0.86486,"outcome":"unexpected_diagnostic","comparisonWallSeconds":0.0213,"maxProgressDeviation":0.0,"longestRepeatedReferenceSeconds":0.0,"dataUrl":"data/real_freeze_middle.json"},{"id":"real_mirror","title":"좌우 반전 · 엄격한 좌우 대응","description":"Horizontally reflect every frame. Strict anatomical correspondence expected.","group":"real_dance","referenceId":"real_floss","candidateId":"real_mirror","expected":"needs_review","expectation_source":"Deterministic transformation stress expectation, not independent correctness ground truth.","status":"unknown","flags":["insufficient_pose_coverage"],"syncMean":0.27073,"dtwMean":0.17777,"coverage":0.78378,"outcome":"inconclusive","comparisonWallSeconds":0.0222,"maxProgressDeviation":0.11111,"longestRepeatedReferenceSeconds":0.33333,"dataUrl":"data/real_mirror.json","mirrorCorrectedMean":0.04482,"mirrorCorrectedStatus":"unknown"},{"id":"real_occluded","title":"중간 구간 영상 소실 · 전체 검정 프레임","description":"Replace middle60% of all pixels with black; missing observations should remain unknown.","group":"real_dance","referenceId":"real_floss","candidateId":"real_occluded","expected":"visibility_loss","expectation_source":"Deterministic transformation stress expectation, not independent correctness ground truth.","status":"unknown","flags":["insufficient_pose_coverage"],"syncMean":0.0166,"dtwMean":0.0166,"coverage":0.35135,"outcome":"observed_visibility_loss","comparisonWallSeconds":0.01842,"maxProgressDeviation":0.0,"longestRepeatedReferenceSeconds":0.0,"dataUrl":"data/real_occluded.json"},{"id":"generated_identity","title":"원본 동일 입력","description":"Exact source copy: extraction determinism check, not an independent test.","group":"generated_controls","referenceId":"generated_reference","candidateId":"generated_identity","expected":"aligned","expectation_source":"Deterministic transformation stress expectation, not independent correctness ground truth.","status":"aligned","flags":[],"syncMean":0.0,"dtwMean":0.0,"coverage":1.0,"outcome":"expected_diagnostic","comparisonWallSeconds":0.10619,"maxProgressDeviation":0.0,"longestRepeatedReferenceSeconds":0.0,"dataUrl":"data/generated_identity.json"},{"id":"generated_slow_135","title":"전체 속도 1.35배 느리게","description":"Whole sequence slowed to1.35x duration with repeated original frames.","group":"generated_controls","referenceId":"generated_reference","candidateId":"generated_slow_135","expected":"aligned","expectation_source":"Deterministic transformation stress expectation, not independent correctness ground truth.","status":"aligned","flags":[],"syncMean":0.00875,"dtwMean":0.00799,"coverage":1.0,"outcome":"expected_diagnostic","comparisonWallSeconds":0.13901,"maxProgressDeviation":0.01134,"longestRepeatedReferenceSeconds":0.25,"dataUrl":"data/generated_slow_135.json"},{"id":"generated_tempo_local","title":"구간별 속도 변화","description":"Nonuniform monotonic retiming: candidate45% corresponds to source22%.","group":"generated_controls","referenceId":"generated_reference","candidateId":"generated_tempo_local","expected":"aligned","expectation_source":"Deterministic transformation stress expectation, not independent correctness ground truth.","status":"aligned","flags":[],"syncMean":0.22594,"dtwMean":0.00958,"coverage":1.0,"outcome":"expected_diagnostic","comparisonWallSeconds":0.10661,"maxProgressDeviation":0.19444,"longestRepeatedReferenceSeconds":0.25,"dataUrl":"data/generated_tempo_local.json"},{"id":"generated_order_swap","title":"동작 구간 순서 교환","description":"Swap the two middle temporal quarters; endpoints preserved.","group":"generated_controls","referenceId":"generated_reference","candidateId":"generated_order_swap","expected":"needs_review","expectation_source":"Deterministic transformation stress expectation, not independent correctness ground truth.","status":"needs_review","flags":["left_arm_tail_deviation","right_arm_tail_deviation","extended_time_warp"],"syncMean":0.16901,"dtwMean":0.08799,"coverage":1.0,"outcome":"expected_diagnostic","comparisonWallSeconds":0.11154,"maxProgressDeviation":0.25,"longestRepeatedReferenceSeconds":1.5,"dataUrl":"data/generated_order_swap.json"},{"id":"generated_skip_middle","title":"중간 동작 생략","description":"Remove source40–65%; no interpolation across the resulting cut.","group":"generated_controls","referenceId":"generated_reference","candidateId":"generated_skip_middle","expected":"needs_review","expectation_source":"Deterministic transformation stress expectation, not independent correctness ground truth.","status":"aligned","flags":[],"syncMean":0.13866,"dtwMean":0.01401,"coverage":1.0,"outcome":"unexpected_diagnostic","comparisonWallSeconds":0.08692,"maxProgressDeviation":0.13194,"longestRepeatedReferenceSeconds":0.0,"dataUrl":"data/generated_skip_middle.json"},{"id":"generated_freeze_middle","title":"중간 프레임 정지","description":"Replace source30–65% with the frame at30%; jump to65% afterward.","group":"generated_controls","referenceId":"generated_reference","candidateId":"generated_freeze_middle","expected":"needs_review","expectation_source":"Deterministic transformation stress expectation, not independent correctness ground truth.","status":"needs_review","flags":["extended_time_warp"],"syncMean":0.11126,"dtwMean":0.06708,"coverage":1.0,"outcome":"expected_diagnostic","comparisonWallSeconds":0.10342,"maxProgressDeviation":0.13194,"longestRepeatedReferenceSeconds":1.33333,"dataUrl":"data/generated_freeze_middle.json"},{"id":"generated_mirror","title":"좌우 반전 · 엄격한 좌우 대응","description":"Horizontally reflect every frame. Strict anatomical correspondence expected.","group":"generated_controls","referenceId":"generated_reference","candidateId":"generated_mirror","expected":"needs_review","expectation_source":"Deterministic transformation stress expectation, not independent correctness ground truth.","status":"needs_review","flags":["left_arm_tail_deviation","right_arm_tail_deviation"],"syncMean":0.13958,"dtwMean":0.13958,"coverage":1.0,"outcome":"expected_diagnostic","comparisonWallSeconds":0.10825,"maxProgressDeviation":0.0,"longestRepeatedReferenceSeconds":0.0,"dataUrl":"data/generated_mirror.json","mirrorCorrectedMean":0.03417,"mirrorCorrectedStatus":"aligned"},{"id":"generated_occluded","title":"중간 구간 영상 소실 · 전체 검정 프레임","description":"Replace middle60% of all pixels with black; missing observations should remain unknown.","group":"generated_controls","referenceId":"generated_reference","candidateId":"generated_occluded","expected":"visibility_loss","expectation_source":"Deterministic transformation stress expectation, not independent correctness ground truth.","status":"unknown","flags":["insufficient_pose_coverage"],"syncMean":0.00377,"dtwMean":0.00377,"coverage":0.4,"outcome":"observed_visibility_loss","comparisonWallSeconds":0.08175,"maxProgressDeviation":0.0,"longestRepeatedReferenceSeconds":0.0,"dataUrl":"data/generated_occluded.json"},{"id":"generated_independent","title":"같은 지시, 별도 생성 영상","description":"순서는 비슷하지만 시점·팔 모양이 다른 두 생성 영상","group":"independent_generated","referenceId":"generated_reference","candidateId":"generated_candidate","expected":"qualitative_review","expectation_source":"Prediction-blind visual draft notes timing and arm-shape differences; no binary expert correctness label.","status":"aligned","flags":[],"syncMean":0.07393,"dtwMean":0.05454,"coverage":1.0,"outcome":"qualitative_only","comparisonWallSeconds":0.10856,"maxProgressDeviation":0.02778,"longestRepeatedReferenceSeconds":0.16667,"dataUrl":"data/generated_independent.json"},{"id":"pose_correct_overhead","title":"양팔 머리 위 · 일치 예시","description":"Both arms overhead; separate generated source","group":"pose_cases","referenceId":"pose_target_overhead","candidateId":"pose_correct_overhead","expected":"aligned","expectation_source":"Prediction-blind assistant visual drafts frozen before pose estimation; no manual angle ground truth.","status":"aligned","flags":[],"syncMean":0.06994,"dtwMean":0.06994,"coverage":1.0,"outcome":"expected_diagnostic","comparisonWallSeconds":0.00956,"maxProgressDeviation":0.0,"longestRepeatedReferenceSeconds":0.0,"dataUrl":"data/pose_correct_overhead.json"},{"id":"pose_wrong_one_arm","title":"한쪽 팔만 머리 위","description":"Only one arm overhead; opposite arm remains down","group":"pose_cases","referenceId":"pose_target_overhead","candidateId":"pose_wrong_one_arm","expected":"needs_review","expectation_source":"Prediction-blind assistant visual drafts frozen before pose estimation; no manual angle ground truth.","status":"needs_review","flags":["left_arm_deviation","left_arm_tail_deviation"],"syncMean":0.31671,"dtwMean":0.31671,"coverage":1.0,"outcome":"expected_diagnostic","comparisonWallSeconds":0.00937,"maxProgressDeviation":0.0,"longestRepeatedReferenceSeconds":0.0,"dataUrl":"data/pose_wrong_one_arm.json"},{"id":"pose_wrong_horizontal","title":"양팔 수평 · 다른 자세","description":"Both arms horizontal instead of overhead","group":"pose_cases","referenceId":"pose_target_overhead","candidateId":"pose_wrong_horizontal","expected":"needs_review","expectation_source":"Prediction-blind assistant visual drafts frozen before pose estimation; no manual angle ground truth.","status":"needs_review","flags":["left_arm_deviation","right_arm_deviation","left_arm_tail_deviation","right_arm_tail_deviation"],"syncMean":0.32694,"dtwMean":0.32694,"coverage":1.0,"outcome":"expected_diagnostic","comparisonWallSeconds":0.0092,"maxProgressDeviation":0.0,"longestRepeatedReferenceSeconds":0.0,"dataUrl":"data/pose_wrong_horizontal.json"},{"id":"pose_wrong_neutral","title":"양팔 아래 · 다른 자세","description":"Both arms down instead of overhead","group":"pose_cases","referenceId":"pose_target_overhead","candidateId":"pose_wrong_neutral","expected":"needs_review","expectation_source":"Prediction-blind assistant visual drafts frozen before pose estimation; no manual angle ground truth.","status":"needs_review","flags":["mean_pose_deviation","left_arm_deviation","right_arm_deviation","left_arm_tail_deviation","right_arm_tail_deviation"],"syncMean":0.50414,"dtwMean":0.50414,"coverage":1.0,"outcome":"expected_diagnostic","comparisonWallSeconds":0.00988,"maxProgressDeviation":0.0,"longestRepeatedReferenceSeconds":0.0,"dataUrl":"data/pose_wrong_neutral.json"}],"groups":[{"id":"pose_cases","title":"명시적 자세 비교","description":"양팔 머리 위 자세를 기준으로 사전에 시각 검토한 1개 일치·3개 불일치 클립"},{"id":"independent_generated","title":"독립 생성 연속 동작","description":"같은 지시로 별도 생성한 두 영상의 관측된 동작을 비교"},{"id":"generated_controls","title":"합성 동작 스트레스 검사","description":"한 합성 원본에서 만든 8개 결정적 변형. 독립 수행자 평가가 아님"},{"id":"real_dance","title":"실제 Floss 춤 스트레스 검사","description":"3.1초 실제 춤 영상과 8개 변형. 반복 춤의 비교 한계를 포함"}],"method":{"name":"YOLO11s-pose + confidence-masked 2D comparison + anchored DTW","config":{"sample_hz":12.0,"band":0.25,"confidence_threshold":0.3,"minimum_coverage":0.8,"mean_distance_review":0.35,"part_distance_review":0.5,"part_p95_review":0.75,"warping_penalty":0.05,"unknown_alignment_cost":1.0,"repeated_assignment_review_s":1.0,"root_motion_review":0.5,"minimum_torso_pixels":1.0,"mirror_candidate":false},"units":"torso_lengths","inference_mode":"Recorded local GPU pose extraction; CPU alignment; static web playback","interpretation":{"status":"aligned means no configured diagnostic flag on sufficiently observed poses; it is not a pass, accuracy, or dance-quality score","coordinate_system":"2D image coordinates, hip-midpoint centered, divided by shoulder-midpoint to hip-midpoint length; no rotation alignment","body_joints":[5,6,7,8,9,10,11,12,13,14,15,16],"resampling":"Nearest actual source frame at 12 Hz by default; no coordinate interpolation. Samples farther than 1.5 nominal source-frame intervals from an observation are unknown. Final endpoints included. Synchronized baseline matches normalized elapsed time, removing only global duration difference.","coverage":"Minimum of source valid-pose coverage, valid alignment-pair fraction, paired body-joint coverage, and each required limb/torso coverage. Below 0.80 is unknown.","missing":"Unknown joints are excluded from observed distances. All four torso anchors are required. Unobservable DTW pairs have optimizer cost 1.0, which is not an observed error.","dtw":"Monotone steps (1,1),(1,0),(0,1), anchored first/last samples, normalized-progress band; non-diagonal cost .05. Order cannot reverse. Repeats can still obscure pauses.","frame_display":"Each candidate frame uses the median reference sample among its path matches, not the lowest-error match. Aggregate DTW metrics use the full path and can differ from displayed-frame averages.","per_part":"Mean/p95 over observed joint distances; coverage and valid must be inspected. Overall mean/p95 summarize valid pair mean distances over body joints5-16.","root_motion":"Secondary pelvis displacement since first observable pelvis, divided by that clip's initial torso length. Not camera-motion invariant. Does not change the root-centered pose distances.","mirror":"Strict left/right correspondence by default; optional explicit candidate reflection swaps COCO left/right indices. The better orientation is never automatically selected.","limitations":["No semantic action recognition or trained action-quality assessment.","Engineering thresholds have not been calibrated on labeled dance outcomes.","2D viewpoint, foreshortening, occlusion and camera motion can dominate errors.","Identity switches from pose extraction are outside this comparator.","Low distances cannot establish that every prescribed semantic stage was performed."]},"paper_scope":"Primary sources only. Availability was checked at paper/repository/documentation level; models and code were not executed by this literature review. No paper reproduction or reported benchmark result is claimed."},"sourceCredits":[{"id":"real_floss","title":"Floss (dance)","author":"LittleT889","url":"https://commons.wikimedia.org/wiki/File:Floss_(dance).gif","license":"CC BY-SA 4.0","license_url":"https://creativecommons.org/licenses/by-sa/4.0/","attribution":"Floss (dance) by LittleT889, via Wikimedia Commons, CC BY-SA 4.0.","changes":"GIF reencoded as H.264; derived cases repeat/retime frames, exchange temporal blocks, omit a block, hold a frame, mirror horizontally, or replace the entire frame with black during the middle60% of the clip. No independent dancer was added.","distribution":"Original and all exported Floss-derived video files are CC BY-SA 4.0. No endorsement implied.","media":["media/real_floss.mp4","media/real_floss.mp4","media/real_slow_135.mp4","media/real_tempo_local.mp4","media/real_order_swap.mp4","media/real_skip_middle.mp4","media/real_freeze_middle.mp4","media/real_mirror.mp4","media/real_occluded.mp4"],"license_file":"media/real-floss-LICENSE.txt"},{"id":"generated","title":"Generated reference, candidate and auxiliary arm sequence","author":"AI-generated with Higgsfield / Seedance 2.5","url":"https://higgsfield.ai/","license":"Generated project media; separate from the CC BY-SA Floss material","attribution":"Higgsfield / Seedance 2.5 generated media for this development experiment.","changes":"Two independently generated 12-second clips, eight deterministic reference variants, and short stable-pose crops from an auxiliary generated sequence.","distribution":"No claim of independently filmed human performance or population-level correctness labels."}],"papers":[{"id":"soft_dtw_2017","tier":"core","title":"Soft-DTW: a Differentiable Loss Function for Time-Series","authors":["Marco Cuturi","Mathieu Blondel"],"year":2017,"venue":"ICML","paper_url":"https://proceedings.mlr.press/v70/cuturi17a.html","code_url":"https://github.com/mblondel/soft-dtw","task":"Differentiable temporal alignment between time series","release_status":"Author paper links official implementation; no pretrained weights needed for the alignment algorithm.","relevance":"Motivates tempo-tolerant alignment. Our planned hard-DTW baseline is not a reproduction of soft-DTW training.","limitations":"Alignment cost alone does not establish correct choreography or preserve required hold durations."},{"id":"tcc_2019","tier":"core","title":"Temporal Cycle-Consistency Learning","authors":["Debidatta Dwibedi","Yusuf Aytar","Jonathan Tompson","Pierre Sermanet","Andrew Zisserman"],"year":2019,"venue":"CVPR","paper_url":"https://openaccess.thecvf.com/content_CVPR_2019/html/Dwibedi_Temporal_Cycle-Consistency_Learning_CVPR_2019_paper.html","code_url":"https://github.com/google-research/google-research/tree/master/tcc","task":"Self-supervised per-frame representation learning and video alignment","release_status":"Official TensorFlow training, embedding extraction, evaluation and alignment visualization code available; README gives ImageNet backbone initialization, not a verified ready-to-use dance compliance checkpoint.","relevance":"Frame correspondences and action phase evaluation; supports showing aligned evidence.","limitations":"Learned alignment is not an instruction-compliance classifier; target-domain validation is needed."},{"id":"lav_2021","tier":"core","title":"Learning by Aligning Videos in Time","authors":["Sanjay Haresh","Sateesh Kumar","Huseyin Coskun","Shahram N. Syed","Andrey Konin","Zeeshan Zia","Quoc-Huy Tran"],"year":2021,"venue":"CVPR","paper_url":"https://arxiv.org/abs/2103.17260","code_url":"https://github.com/trquhuytin/LAV-CVPR21","project_url":"https://retrocausal.ai/learning-by-aligning-videos-in-time/","task":"Self-supervised video representation learning using soft-DTW and temporal regularization","release_status":"Official PyTorch code, environment, train/evaluate commands available; no trained dance checkpoint identified in the checked README.","relevance":"Explains why temporal alignment and temporal discrimination must be considered together.","limitations":"Paper is CVPR 2021, not 2022. Evaluates Pouring, Penn Action, IKEA ASM; does not directly certify dance imitation correctness."},{"id":"pr_vipe_2020","tier":"core","title":"View-Invariant Probabilistic Embedding for Human Pose","authors":["Jennifer J. Sun","Jiaping Zhao","Liang-Chieh Chen","Florian Schroff","Hartwig Adam","Ting Liu"],"year":2020,"venue":"ECCV","paper_url":"https://arxiv.org/abs/1912.01001","code_url":"https://github.com/google-research/google-research/tree/master/poem/pr_vipe","project_url":"https://sites.google.com/view/pr-vipe/home","task":"View-invariant probabilistic embeddings from 2D joint keypoints","release_status":"Official TensorFlow training and infer.py available. Author discussion confirms an original pretrained checkpoint existed; checkpoint download was not tested in this review.","relevance":"Alternative representation when camera viewpoint varies.","limitations":"Author notes checkpoint sensitivity to keypoint-detector domain and original checkpoint missing-keypoint limitations. Embedding similarity does not localize anatomical errors by itself."},{"id":"blazepose_2020","tier":"core","title":"BlazePose: On-device Real-time Body Pose tracking","authors":["Valentin Bazarevsky","Ivan Grishchenko","Karthik Raveendran","Tyler Zhu","Fan Zhang","Matthias Grundmann"],"year":2020,"venue":"arXiv; CVPR CV4ARVR workshop presentation","paper_url":"https://arxiv.org/abs/2006.10204","task":"Body pose detection and tracking","release_status":"Current official MediaPipe Tasks docs provide Python/Web guides and downloadable Lite/Full/Heavy pose-landmarker bundles.","relevance":"Practical single-person RGB pose frontend; confidence-gated normalized keypoints are inspectable.","limitations":"Current MediaPipe bundle is a BlazePose/GHUM variant, not an exact reproduction of the 2020 architecture. Monocular inferred world coordinates are not motion-capture ground truth."},{"id":"rtmpose_2023","tier":"core","title":"RTMPose: Real-Time Multi-Person Pose Estimation based on MMPose","authors":["Tao Jiang","Peng Lu","Li Zhang","Ningsheng Ma","Rui Han","Chengqi Lyu","Yining Li","Kai Chen"],"year":2023,"venue":"arXiv","paper_url":"https://arxiv.org/abs/2303.07399","code_url":"https://github.com/open-mmlab/mmpose/tree/main/projects/rtmpose","task":"Real-time 2D pose estimation","release_status":"Official model zoo links PTH/ONNX weights and inference examples. Official README also recommends rtmlib for ONNXRuntime/OpenCV/OpenVINO without PyTorch/MMCV.","relevance":"Alternative pose frontend, including whole-body variants and explicit deployment choices.","limitations":"Top-down pipeline needs person localization; pose confidence/AP is not action-compliance accuracy. Author FPS cannot be substituted for our full-pipeline timing."},{"id":"aistpp_2021","tier":"core","title":"AI Choreographer: Music Conditioned 3D Dance Generation with AIST++","authors":["Ruilong Li","Shan Yang","David A. Ross","Angjoo Kanazawa"],"year":2021,"venue":"ICCV","paper_url":"https://arxiv.org/abs/2101.08779","code_url":"https://github.com/google/aistplusplus_api","project_url":"https://google.github.io/aistplusplus_dataset/","task":"Dance dataset and music-conditioned 3D dance generation","release_status":"Official annotation downloads and API available; video downloader points to original AIST videos and their terms. SMPL-related assets have separate requirements.","relevance":"Real dance motion/camera data and cross-view tests; reference and candidate selection can be made explicit.","limitations":"1,408 sequences and 10 genres describe dataset scope, not imitation-correctness labels. Official docs require exact 60-FPS frame extraction and recommend an ignore list for poor reconstructions."},{"id":"finedance_2023","tier":"core","title":"FineDance: A Fine-grained Choreography Dataset for 3D Full Body Dance Generation","authors":["Ronghui Li","Junfan Zhao","Yachao Zhang","Mingyang Su","Zeping Ren","Han Zhang","Yansong Tang","Xiu Li"],"year":2023,"venue":"ICCV","paper_url":"https://arxiv.org/abs/2212.03741","code_url":"https://github.com/li-ronghui/FineDance","task":"Full-body dance generation and music-motion data","release_status":"Official repo links a 7.7-hour public subset, pretrained generation checkpoints/assets and inference scripts.","relevance":"Dance-specific motions and dancer-disjoint evaluation design; useful future benchmark source.","limitations":"Paper describes 14.6 hours, while checked release states a 7.7-hour subset. Generation checkpoint is not a reference-imitation assessor; SMPLH data and assets need their own setup."},{"id":"llamo_2025","tier":"core","title":"Human Motion Instruction Tuning","authors":["Lei Li","Sen Jia","Jianhao Wang","Zhongyu Jiang","Feng Zhou","Ju Dai","Tianfang Zhang","Zongkai Wu","Jenq-Neng Hwang"],"year":2025,"venue":"CVPR","paper_url":"https://openaccess.thecvf.com/content/CVPR2025/html/Li_Human_Motion_Instruction_Tuning_CVPR_2025_paper.html","code_url":"https://github.com/ILGLJ/LLaMo","task":"Video, native motion and text conditioned human-motion understanding","release_status":"Paper says code/models available, but checked official repository contained only README.md and no runnable implementation or checkpoint link.","relevance":"Relevant direction for natural-language motion questions and descriptions.","limitations":"Instruction tuning is not evidence of calibrated arbitrary instruction-compliance decisions; not selected for the runnable baseline."},{"id":"dancematch_2026","tier":"core","title":"Learning Quantised Structure-Preserving Motion Representations for Dance Fingerprinting","authors":["Arina Kharlamova","Bowei He","Chen Ma","Xue Liu"],"year":2026,"venue":"arXiv preprint, submitted 2026-04-01","paper_url":"https://arxiv.org/abs/2604.00927","task":"Motion-based dance retrieval","release_status":"Paper claims a dataset release; no author code, dataset download or pretrained checkpoint URL was identified in the checked abstract/full text/search.","relevance":"Recent example distinguishing global dance retrieval from local correctness and temporal order.","limitations":"Retrieval accuracy is not imitation-quality accuracy. The paper explicitly lists loss of temporal ordering among limitations; no reproduction claimed."},{"id":"fine_diving_2022","tier":"related_quality","title":"FineDiving: A Fine-grained Dataset for Procedure-aware Action Quality Assessment","authors":["Jinglin Xu","Yongming Rao","Xumin Yu","Guangyi Chen","Jie Zhou","Jiwen Lu"],"year":2022,"venue":"CVPR","paper_url":"https://openaccess.thecvf.com/content/CVPR2022/papers/Xu_FineDiving_A_Fine-Grained_Dataset_for_Procedure-Aware_Action_Quality_Assessment_CVPR_2022_paper.pdf","code_url":"https://github.com/xujinglin/FineDiving","task":"Procedure-aware diving quality assessment","release_status":"Official TSA training/evaluation code available; dataset requires a signed agreement and email request. README test checkpoint is produced by training.","relevance":"Supports explicit sub-action phases and reference/query comparisons.","limitations":"Diving scoring and known step-transition count assumptions do not transfer directly to open dance imitation."},{"id":"fineparser_2024","tier":"related_quality","title":"FineParser: A Fine-grained Spatio-temporal Action Parser for Human-centric Action Quality Assessment","authors":["Jinglin Xu","Sibo Yin","Guohao Zhao","Zishuo Wang","Yuxin Peng"],"year":2024,"venue":"CVPR","paper_url":"https://arxiv.org/abs/2405.06887","code_url":"https://github.com/PKU-ICST-MIPL/FineParser_CVPR2024","task":"Human-centric spatiotemporal parsing for diving quality assessment","release_status":"Official PyTorch code and train/test configuration available. FineDiving-HM requires release agreement/email; README links I3D backbone weights, not a verified final fine-tuned FineParser checkpoint.","relevance":"Fine-grained spatial and temporal evidence is more useful than one opaque scalar.","limitations":"Diving-domain supervision and masks are required; no arbitrary choreography pass/fail transfer established."},{"id":"nsaqa_2024","tier":"related_quality","title":"Hierarchical NeuroSymbolic Approach for Comprehensive and Explainable Action Quality Assessment","authors":["Lauren Okamoto","Paritosh Parmar"],"year":2024,"venue":"CVPR Workshops, CVsports","paper_url":"https://arxiv.org/abs/2403.13798","code_url":"https://github.com/laurenok24/NSAQA","task":"Neural symbols plus rules for explainable platform-diving assessment","release_status":"Official Python single-video entrypoint and HTML report code, plus hosted demo link. Dependency/weight completeness not executed or verified; repo restricts materials to non-commercial use.","relevance":"Conceptual support for explainable evidence plus an explicit decision specification.","limitations":"The rules are sport-specific. This experiment borrows the separation-of-evidence-and-rules idea, not its code or trained pipeline."},{"id":"fitness_aqa_2022","tier":"related_quality","title":"Domain Knowledge-Informed Self-Supervised Representations for Workout Form Assessment","authors":["Paritosh Parmar","Amol Gharat","Helge Rhodin"],"year":2022,"venue":"ECCV","paper_url":"https://arxiv.org/abs/2202.14019","code_url":"https://github.com/ParitoshParmar/Fitness-AQA","task":"Workout form assessment for back squat, overhead press and barbell row","release_status":"Code_Release directory available; dataset is request-based and non-commercial. Turnkey pretrained inference not verified.","relevance":"Demonstrates that recognizing an exercise and detecting execution errors are separate objectives.","limitations":"Exercise-specific expert error labels do not provide dance-sequence correctness labels."},{"id":"flex_2025","tier":"related_quality","title":"FLEX: A Large-Scale Multi-Modal Multi-Action Dataset for Fitness Action Quality Assessment","authors":["Hao Yin","Lijun Gu","Paritosh Parmar","Lin Xu","Tianxiao Guo","Weiwei Fu","Yang Zhang","Tianyou Zheng"],"year":2025,"venue":"arXiv preprint","paper_url":"https://arxiv.org/abs/2506.03198","project_url":"https://haoyin116.github.io/FLEX_Dataset","task":"Multimodal fitness action quality dataset with structured error feedback","release_status":"Paper advertises code/data; linked project returned HTTP 404 during this review, so immediate download/weights could not be verified.","relevance":"Structured errors and feedback offer a better target than a generic action label.","limitations":"Weight-loaded fitness, 3D pose and sEMG setting differs from monocular dance imitation. An ICLR 2026 submission PDF was seen, but acceptance was not verified."},{"id":"posec3d_2022","tier":"related_recognition","title":"Revisiting Skeleton-Based Action Recognition","authors":["Haodong Duan","Yue Zhao","Kai Chen","Dahua Lin","Bo Dai"],"year":2022,"venue":"CVPR","paper_url":"https://openaccess.thecvf.com/content/CVPR2022/html/Duan_Revisiting_Skeleton-Based_Action_Recognition_CVPR_2022_paper.html","code_url":"https://github.com/kennymckormick/pyskl","task":"Action recognition from spatiotemporal pose heatmaps","release_status":"Official repository and PoseC3D configs/training/testing instructions available; not installed in this review.","relevance":"Useful recognition baseline and pose representation alternative.","limitations":"Correct action class does not establish correct side, amplitude, phrase order or completion."}],"summary":{"planned_cases":21,"completed_cases":21,"complete":true,"status_counts":{"aligned":12,"needs_review":6,"unknown":3},"pose_confusion":{"cases":4,"expected_positive":1,"expected_negative":3,"counts":{"tp":1,"fn":0,"fp":0,"tn":3,"unknown_positive":0,"unknown_negative":0},"matrix":{"aligned":{"aligned":1,"needs_review":0,"unknown":0},"needs_review":{"aligned":0,"needs_review":3,"unknown":0}},"positive_definition":"Coarse visual match to both-arms-overhead target, not exact joint angles.","interpretation":"One positive and three negative assistant-reviewed draft cases from related generated clips. Unknown is an abstention. No population accuracy estimate."},"limitations":["Development stress cases share their source; not 21 independent dancers.","Four posture labels are prediction-blind assistant drafts, not expert angle ground truth.","Aligned is a diagnostic state, not a validated correct-performance judgment.","Pipeline is 2D pose+DTW, not a reproduction of a learned paper model."],"unexpected_cases":["real_order_swap","real_skip_middle","real_freeze_middle","generated_skip_middle"],"inconclusive_cases":["real_mirror"],"paper_count":16,"unique_video_assets":22},"assets":{"real_floss":{"id":"real_floss","video":"media/real_floss.mp4","fps":10.0,"width":292,"height":528,"duration":3.1,"videoSha256":"fd42cf357eb6e311b1ae3f17241852c429f9e845856410ecfe22ef6e7a4b2619","poseRuntime":{"device":"NVIDIA GeForce RTX 4090","precision":"float32","input_size":960,"model":"yolo11s-pose.pt","model_sha256":"1060bda4a27012060eca246f9b2adeea22eabb045a1e58f8d229be29b7ebc2ba","mean_ms":10.19196,"p95_ms":14.2426,"inverse_mean_fps":98.11657,"timing_policy":"3 warmups, CUDA-synchronized wall time; decode, pose, postprocess included. JSON writes and visualization excluded.","valid_person_frames":30,"selection_policy":"Exactly one detected person. Zero or multiple people => unknown; no inferred keypoints filled."}},"real_identity":{"id":"real_identity","video":"media/real_floss.mp4","fps":10.0,"width":292,"height":528,"duration":3.1,"videoSha256":"fd42cf357eb6e311b1ae3f17241852c429f9e845856410ecfe22ef6e7a4b2619","poseRuntime":{"device":"NVIDIA GeForce RTX 4090","precision":"float32","input_size":960,"model":"yolo11s-pose.pt","model_sha256":"1060bda4a27012060eca246f9b2adeea22eabb045a1e58f8d229be29b7ebc2ba","mean_ms":9.82136,"p95_ms":12.9322,"inverse_mean_fps":101.81891,"timing_policy":"3 warmups, CUDA-synchronized wall time; decode, pose, postprocess included. JSON writes and visualization excluded.","valid_person_frames":30,"selection_policy":"Exactly one detected person. Zero or multiple people => unknown; no inferred keypoints filled."}},"real_slow_135":{"id":"real_slow_135","video":"media/real_slow_135.mp4","fps":10.0,"width":292,"height":528,"duration":4.2,"videoSha256":"044200b6d92d2be1d529ca5d1ba504512fb3bb29a515dedcbd85e5135db85716","poseRuntime":{"device":"NVIDIA GeForce RTX 4090","precision":"float32","input_size":960,"model":"yolo11s-pose.pt","model_sha256":"1060bda4a27012060eca246f9b2adeea22eabb045a1e58f8d229be29b7ebc2ba","mean_ms":10.58502,"p95_ms":13.93714,"inverse_mean_fps":94.4731,"timing_policy":"3 warmups, CUDA-synchronized wall time; decode, pose, postprocess included. JSON writes and visualization excluded.","valid_person_frames":38,"selection_policy":"Exactly one detected person. Zero or multiple people => unknown; no inferred keypoints filled."}},"real_tempo_local":{"id":"real_tempo_local","video":"media/real_tempo_local.mp4","fps":10.0,"width":292,"height":528,"duration":3.1,"videoSha256":"22d5414baeaae34b47561d914cab4fdb3a8f1961f9011d98605414717ff9c761","poseRuntime":{"device":"NVIDIA GeForce RTX 4090","precision":"float32","input_size":960,"model":"yolo11s-pose.pt","model_sha256":"1060bda4a27012060eca246f9b2adeea22eabb045a1e58f8d229be29b7ebc2ba","mean_ms":10.33991,"p95_ms":14.67645,"inverse_mean_fps":96.71264,"timing_policy":"3 warmups, CUDA-synchronized wall time; decode, pose, postprocess included. JSON writes and visualization excluded.","valid_person_frames":28,"selection_policy":"Exactly one detected person. Zero or multiple people => unknown; no inferred keypoints filled."}},"real_order_swap":{"id":"real_order_swap","video":"media/real_order_swap.mp4","fps":10.0,"width":292,"height":528,"duration":3.1,"videoSha256":"1badcb5ff1e4f142a90aee871d2b991811db1cd3ac92517f637e7174a52b441a","poseRuntime":{"device":"NVIDIA GeForce RTX 4090","precision":"float32","input_size":960,"model":"yolo11s-pose.pt","model_sha256":"1060bda4a27012060eca246f9b2adeea22eabb045a1e58f8d229be29b7ebc2ba","mean_ms":9.93882,"p95_ms":12.6576,"inverse_mean_fps":100.61561,"timing_policy":"3 warmups, CUDA-synchronized wall time; decode, pose, postprocess included. JSON writes and visualization excluded.","valid_person_frames":27,"selection_policy":"Exactly one detected person. Zero or multiple people => unknown; no inferred keypoints filled."}},"real_skip_middle":{"id":"real_skip_middle","video":"media/real_skip_middle.mp4","fps":10.0,"width":292,"height":528,"duration":2.3,"videoSha256":"6fc44e7c5535f6c73b1b399087ba171eeb4954cca815d094ebf1e52dbdf6a812","poseRuntime":{"device":"NVIDIA GeForce RTX 4090","precision":"float32","input_size":960,"model":"yolo11s-pose.pt","model_sha256":"1060bda4a27012060eca246f9b2adeea22eabb045a1e58f8d229be29b7ebc2ba","mean_ms":9.53149,"p95_ms":11.38758,"inverse_mean_fps":104.91542,"timing_policy":"3 warmups, CUDA-synchronized wall time; decode, pose, postprocess included. JSON writes and visualization excluded.","valid_person_frames":19,"selection_policy":"Exactly one detected person. Zero or multiple people => unknown; no inferred keypoints filled."}},"real_freeze_middle":{"id":"real_freeze_middle","video":"media/real_freeze_middle.mp4","fps":10.0,"width":292,"height":528,"duration":3.1,"videoSha256":"0074e6ff2c6fb019a5cec0b066093e47a4b4ec9ae27a7d5d2b25392be83651f3","poseRuntime":{"device":"NVIDIA GeForce RTX 4090","precision":"float32","input_size":960,"model":"yolo11s-pose.pt","model_sha256":"1060bda4a27012060eca246f9b2adeea22eabb045a1e58f8d229be29b7ebc2ba","mean_ms":9.82997,"p95_ms":12.3476,"inverse_mean_fps":101.72973,"timing_policy":"3 warmups, CUDA-synchronized wall time; decode, pose, postprocess included. JSON writes and visualization excluded.","valid_person_frames":27,"selection_policy":"Exactly one detected person. Zero or multiple people => unknown; no inferred keypoints filled."}},"real_mirror":{"id":"real_mirror","video":"media/real_mirror.mp4","fps":10.0,"width":292,"height":528,"duration":3.1,"videoSha256":"2b484cba4ffb885547156308967195158960ebccd1ffa48790d2d3f38404b8f6","poseRuntime":{"device":"NVIDIA GeForce RTX 4090","precision":"float32","input_size":960,"model":"yolo11s-pose.pt","model_sha256":"1060bda4a27012060eca246f9b2adeea22eabb045a1e58f8d229be29b7ebc2ba","mean_ms":9.78696,"p95_ms":13.5866,"inverse_mean_fps":102.17679,"timing_policy":"3 warmups, CUDA-synchronized wall time; decode, pose, postprocess included. JSON writes and visualization excluded.","valid_person_frames":25,"selection_policy":"Exactly one detected person. Zero or multiple people => unknown; no inferred keypoints filled."}},"real_occluded":{"id":"real_occluded","video":"media/real_occluded.mp4","fps":10.0,"width":292,"height":528,"duration":3.1,"videoSha256":"218efebd601faa6eef449c7ed3a7fee772d32c668ca4211a072fa3702ad37f62","poseRuntime":{"device":"NVIDIA GeForce RTX 4090","precision":"float32","input_size":960,"model":"yolo11s-pose.pt","model_sha256":"1060bda4a27012060eca246f9b2adeea22eabb045a1e58f8d229be29b7ebc2ba","mean_ms":9.1706,"p95_ms":12.62145,"inverse_mean_fps":109.04416,"timing_policy":"3 warmups, CUDA-synchronized wall time; decode, pose, postprocess included. JSON writes and visualization excluded.","valid_person_frames":11,"selection_policy":"Exactly one detected person. Zero or multiple people => unknown; no inferred keypoints filled."}},"generated_reference":{"id":"generated_reference","video":"media/generated_reference.mp4","fps":24.0,"width":1280,"height":720,"duration":12.04167,"videoSha256":"b2683b717df3ca0685de11c707f1ea80ce1a5b099c51892b86da631733546201","poseRuntime":{"device":"NVIDIA GeForce RTX 4090","precision":"float32","input_size":960,"model":"yolo11s-pose.pt","model_sha256":"1060bda4a27012060eca246f9b2adeea22eabb045a1e58f8d229be29b7ebc2ba","mean_ms":12.12275,"p95_ms":17.82254,"inverse_mean_fps":82.48954,"timing_policy":"3 warmups, CUDA-synchronized wall time; decode, pose, postprocess included. JSON writes and visualization excluded.","valid_person_frames":289,"selection_policy":"Exactly one detected person. Zero or multiple people => unknown; no inferred keypoints filled."}},"generated_identity":{"id":"generated_identity","video":"media/generated_reference.mp4","fps":24.0,"width":1280,"height":720,"duration":12.04167,"videoSha256":"b2683b717df3ca0685de11c707f1ea80ce1a5b099c51892b86da631733546201","poseRuntime":{"device":"NVIDIA GeForce RTX 4090","precision":"float32","input_size":960,"model":"yolo11s-pose.pt","model_sha256":"1060bda4a27012060eca246f9b2adeea22eabb045a1e58f8d229be29b7ebc2ba","mean_ms":11.83279,"p95_ms":16.721,"inverse_mean_fps":84.51089,"timing_policy":"3 warmups, CUDA-synchronized wall time; decode, pose, postprocess included. JSON writes and visualization excluded.","valid_person_frames":289,"selection_policy":"Exactly one detected person. Zero or multiple people => unknown; no inferred keypoints filled."}},"generated_slow_135":{"id":"generated_slow_135","video":"media/generated_slow_135.mp4","fps":24.0,"width":1280,"height":720,"duration":16.25,"videoSha256":"57a12249a895a32dd8d8b966ce6e5814779c76b56e161aed8ebfde9731680acb","poseRuntime":{"device":"NVIDIA GeForce RTX 4090","precision":"float32","input_size":960,"model":"yolo11s-pose.pt","model_sha256":"1060bda4a27012060eca246f9b2adeea22eabb045a1e58f8d229be29b7ebc2ba","mean_ms":12.49319,"p95_ms":18.22397,"inverse_mean_fps":80.04364,"timing_policy":"3 warmups, CUDA-synchronized wall time; decode, pose, postprocess included. JSON writes and visualization excluded.","valid_person_frames":390,"selection_policy":"Exactly one detected person. Zero or multiple people => unknown; no inferred keypoints filled."}},"generated_tempo_local":{"id":"generated_tempo_local","video":"media/generated_tempo_local.mp4","fps":24.0,"width":1280,"height":720,"duration":12.04167,"videoSha256":"4a939235a9d206f6fc094119c8acecee6fb030de643c81508cbfaa7b85745d3e","poseRuntime":{"device":"NVIDIA GeForce RTX 4090","precision":"float32","input_size":960,"model":"yolo11s-pose.pt","model_sha256":"1060bda4a27012060eca246f9b2adeea22eabb045a1e58f8d229be29b7ebc2ba","mean_ms":12.31665,"p95_ms":17.52184,"inverse_mean_fps":81.19094,"timing_policy":"3 warmups, CUDA-synchronized wall time; decode, pose, postprocess included. JSON writes and visualization excluded.","valid_person_frames":289,"selection_policy":"Exactly one detected person. Zero or multiple people => unknown; no inferred keypoints filled."}},"generated_order_swap":{"id":"generated_order_swap","video":"media/generated_order_swap.mp4","fps":24.0,"width":1280,"height":720,"duration":12.04167,"videoSha256":"89a96001c82034630b8f57f67440d650f494d59c3b1e3c83e5d5d60856edd7ac","poseRuntime":{"device":"NVIDIA GeForce RTX 4090","precision":"float32","input_size":960,"model":"yolo11s-pose.pt","model_sha256":"1060bda4a27012060eca246f9b2adeea22eabb045a1e58f8d229be29b7ebc2ba","mean_ms":11.69153,"p95_ms":16.5305,"inverse_mean_fps":85.532,"timing_policy":"3 warmups, CUDA-synchronized wall time; decode, pose, postprocess included. JSON writes and visualization excluded.","valid_person_frames":289,"selection_policy":"Exactly one detected person. Zero or multiple people => unknown; no inferred keypoints filled."}},"generated_skip_middle":{"id":"generated_skip_middle","video":"media/generated_skip_middle.mp4","fps":24.0,"width":1280,"height":720,"duration":9.04167,"videoSha256":"4091af457adc0c32cc5d1c1f3696f97a4084996c2c93421fe691023e51cfb0da","poseRuntime":{"device":"NVIDIA GeForce RTX 4090","precision":"float32","input_size":960,"model":"yolo11s-pose.pt","model_sha256":"1060bda4a27012060eca246f9b2adeea22eabb045a1e58f8d229be29b7ebc2ba","mean_ms":11.49231,"p95_ms":15.40692,"inverse_mean_fps":87.01468,"timing_policy":"3 warmups, CUDA-synchronized wall time; decode, pose, postprocess included. JSON writes and visualization excluded.","valid_person_frames":217,"selection_policy":"Exactly one detected person. Zero or multiple people => unknown; no inferred keypoints filled."}},"generated_freeze_middle":{"id":"generated_freeze_middle","video":"media/generated_freeze_middle.mp4","fps":24.0,"width":1280,"height":720,"duration":12.04167,"videoSha256":"a1f8f059cdbbe696631baa6f340611093a7737bd29e2b70385d6c40a54109de3","poseRuntime":{"device":"NVIDIA GeForce RTX 4090","precision":"float32","input_size":960,"model":"yolo11s-pose.pt","model_sha256":"1060bda4a27012060eca246f9b2adeea22eabb045a1e58f8d229be29b7ebc2ba","mean_ms":13.04988,"p95_ms":19.7248,"inverse_mean_fps":76.62906,"timing_policy":"3 warmups, CUDA-synchronized wall time; decode, pose, postprocess included. JSON writes and visualization excluded.","valid_person_frames":289,"selection_policy":"Exactly one detected person. Zero or multiple people => unknown; no inferred keypoints filled."}},"generated_mirror":{"id":"generated_mirror","video":"media/generated_mirror.mp4","fps":24.0,"width":1280,"height":720,"duration":12.04167,"videoSha256":"6074ec94e742ceef1a19e6d0621875687ca3e4a169806d4bf3ae600466d6925e","poseRuntime":{"device":"NVIDIA GeForce RTX 4090","precision":"float32","input_size":960,"model":"yolo11s-pose.pt","model_sha256":"1060bda4a27012060eca246f9b2adeea22eabb045a1e58f8d229be29b7ebc2ba","mean_ms":12.17553,"p95_ms":17.24734,"inverse_mean_fps":82.13196,"timing_policy":"3 warmups, CUDA-synchronized wall time; decode, pose, postprocess included. JSON writes and visualization excluded.","valid_person_frames":289,"selection_policy":"Exactly one detected person. Zero or multiple people => unknown; no inferred keypoints filled."}},"generated_occluded":{"id":"generated_occluded","video":"media/generated_occluded.mp4","fps":24.0,"width":1280,"height":720,"duration":12.04167,"videoSha256":"b5085d4ccc827c53dc898ed94d58ccb23170a19f5b85b5fd322a33714e9611d8","poseRuntime":{"device":"NVIDIA GeForce RTX 4090","precision":"float32","input_size":960,"model":"yolo11s-pose.pt","model_sha256":"1060bda4a27012060eca246f9b2adeea22eabb045a1e58f8d229be29b7ebc2ba","mean_ms":11.29384,"p95_ms":16.09986,"inverse_mean_fps":88.54383,"timing_policy":"3 warmups, CUDA-synchronized wall time; decode, pose, postprocess included. JSON writes and visualization excluded.","valid_person_frames":115,"selection_policy":"Exactly one detected person. Zero or multiple people => unknown; no inferred keypoints filled."}},"generated_candidate":{"id":"generated_candidate","video":"media/generated_candidate.mp4","fps":24.0,"width":1280,"height":720,"duration":12.04167,"videoSha256":"b05ec4c5dc4359b7ac62d7341f654aad9e1c7323684a69be95b698a66f2b5109","poseRuntime":{"device":"NVIDIA GeForce RTX 4090","precision":"float32","input_size":960,"model":"yolo11s-pose.pt","model_sha256":"1060bda4a27012060eca246f9b2adeea22eabb045a1e58f8d229be29b7ebc2ba","mean_ms":11.65282,"p95_ms":15.86416,"inverse_mean_fps":85.81613,"timing_policy":"3 warmups, CUDA-synchronized wall time; decode, pose, postprocess included. JSON writes and visualization excluded.","valid_person_frames":289,"selection_policy":"Exactly one detected person. Zero or multiple people => unknown; no inferred keypoints filled."}},"pose_target_overhead":{"id":"pose_target_overhead","video":"media/pose_target_overhead.mp4","fps":24.0,"width":1280,"height":720,"duration":1.0,"videoSha256":"a8929cf7c5cb72092ba26ec712ba56d62d6f6b91ad0148c83122fc3b5ed4af6a","poseRuntime":{"device":"NVIDIA GeForce RTX 4090","precision":"float32","input_size":960,"model":"yolo11s-pose.pt","model_sha256":"1060bda4a27012060eca246f9b2adeea22eabb045a1e58f8d229be29b7ebc2ba","mean_ms":11.74645,"p95_ms":17.4108,"inverse_mean_fps":85.1321,"timing_policy":"3 warmups, CUDA-synchronized wall time; decode, pose, postprocess included. JSON writes and visualization excluded.","valid_person_frames":24,"selection_policy":"Exactly one detected person. Zero or multiple people => unknown; no inferred keypoints filled."}},"pose_correct_overhead":{"id":"pose_correct_overhead","video":"media/pose_correct_overhead.mp4","fps":24.0,"width":1280,"height":720,"duration":1.0,"videoSha256":"02ce31a259d4721cf40c7035804905b46f211f08e66efc5320279c09f08f77fe","poseRuntime":{"device":"NVIDIA GeForce RTX 4090","precision":"float32","input_size":960,"model":"yolo11s-pose.pt","model_sha256":"1060bda4a27012060eca246f9b2adeea22eabb045a1e58f8d229be29b7ebc2ba","mean_ms":12.01631,"p95_ms":15.03983,"inverse_mean_fps":83.22023,"timing_policy":"3 warmups, CUDA-synchronized wall time; decode, pose, postprocess included. JSON writes and visualization excluded.","valid_person_frames":24,"selection_policy":"Exactly one detected person. Zero or multiple people => unknown; no inferred keypoints filled."}},"pose_wrong_one_arm":{"id":"pose_wrong_one_arm","video":"media/pose_wrong_one_arm.mp4","fps":24.0,"width":1280,"height":720,"duration":1.0,"videoSha256":"30c36140cd396ac5e2b8533711d13f33a26c46dab0515b316b50ded33414c75b","poseRuntime":{"device":"NVIDIA GeForce RTX 4090","precision":"float32","input_size":960,"model":"yolo11s-pose.pt","model_sha256":"1060bda4a27012060eca246f9b2adeea22eabb045a1e58f8d229be29b7ebc2ba","mean_ms":11.7949,"p95_ms":15.94768,"inverse_mean_fps":84.78244,"timing_policy":"3 warmups, CUDA-synchronized wall time; decode, pose, postprocess included. JSON writes and visualization excluded.","valid_person_frames":24,"selection_policy":"Exactly one detected person. Zero or multiple people => unknown; no inferred keypoints filled."}},"pose_wrong_horizontal":{"id":"pose_wrong_horizontal","video":"media/pose_wrong_horizontal.mp4","fps":24.0,"width":1280,"height":720,"duration":1.0,"videoSha256":"37597bb644587d5c6982c738c7601e9a620319be320ca54f836687780e26fc81","poseRuntime":{"device":"NVIDIA GeForce RTX 4090","precision":"float32","input_size":960,"model":"yolo11s-pose.pt","model_sha256":"1060bda4a27012060eca246f9b2adeea22eabb045a1e58f8d229be29b7ebc2ba","mean_ms":14.93101,"p95_ms":22.9025,"inverse_mean_fps":66.97469,"timing_policy":"3 warmups, CUDA-synchronized wall time; decode, pose, postprocess included. JSON writes and visualization excluded.","valid_person_frames":24,"selection_policy":"Exactly one detected person. Zero or multiple people => unknown; no inferred keypoints filled."}},"pose_wrong_neutral":{"id":"pose_wrong_neutral","video":"media/pose_wrong_neutral.mp4","fps":24.0,"width":1280,"height":720,"duration":1.0,"videoSha256":"e1be5b79e98fb95bb4fd7e32f85805f20136dde63e3992f61aee711448e63284","poseRuntime":{"device":"NVIDIA GeForce RTX 4090","precision":"float32","input_size":960,"model":"yolo11s-pose.pt","model_sha256":"1060bda4a27012060eca246f9b2adeea22eabb045a1e58f8d229be29b7ebc2ba","mean_ms":11.51263,"p95_ms":14.5759,"inverse_mean_fps":86.86116,"timing_policy":"3 warmups, CUDA-synchronized wall time; decode, pose, postprocess included. JSON writes and visualization excluded.","valid_person_frames":24,"selection_policy":"Exactly one detected person. Zero or multiple people => unknown; no inferred keypoints filled."}}}}