Files
modeling_zhaocui/final/output/q1/alignment_query_example.json

49 lines
1.3 KiB
JSON

{
"sample_id": "-iRBcNs9oI8/8",
"selection": "manually selected example with valid audio and vision query maps",
"duration_s": 8.093000411987305,
"word_count": 21,
"words": [
"I've",
"drawn",
"this",
"man",
"here,",
"and",
"if",
"you",
"can",
"tell,",
"he",
"has",
"a",
"lot",
"of",
"big",
"muscles,",
"he",
"looks",
"very",
"strong"
],
"audio_query_shape": [
21,
803
],
"audio_query_nnz": 495,
"audio_query_mapped_words": 21,
"vision_query_shape": [
21,
241
],
"vision_query_nnz": 160,
"vision_query_mapped_words": 21,
"audio_query_feature_validity_channel": "log_energy (audio feature dimension 66)",
"vision_query_feature_validity_channel": "OpenFace AU12 intensity (vision feature dimension 10)",
"alignment_semantics": "per-word normalized overlap between stored CTC word intervals and native source intervals; no learned content similarity",
"coordinate_axes_shown": false,
"color_scale": "shared square-root intensity transform for visibility; original overlap weights retained",
"word_label_positions": "actual transcript labels placed at their CTC word-interval centers on the source-time axis",
"image": "alignment_query_example.png"
}