{ "sample_id": "-iRBcNs9oI8/8", "selection": "manually selected example with valid audio and vision query maps", "duration_s": 8.093000411987305, "word_count": 21, "words": [ "I've", "drawn", "this", "man", "here,", "and", "if", "you", "can", "tell,", "he", "has", "a", "lot", "of", "big", "muscles,", "he", "looks", "very", "strong" ], "audio_query_shape": [ 21, 803 ], "audio_query_nnz": 495, "audio_query_mapped_words": 21, "vision_query_shape": [ 21, 241 ], "vision_query_nnz": 160, "vision_query_mapped_words": 21, "audio_query_feature_validity_channel": "log_energy (audio feature dimension 66)", "vision_query_feature_validity_channel": "OpenFace AU12 intensity (vision feature dimension 10)", "alignment_semantics": "per-word normalized overlap between stored CTC word intervals and native source intervals; no learned content similarity", "coordinate_axes_shown": false, "color_scale": "shared square-root intensity transform for visibility; original overlap weights retained", "word_label_positions": "actual transcript labels placed at their CTC word-interval centers on the source-time axis", "image": "alignment_query_example.png" }