49 lines
1.3 KiB
JSON
49 lines
1.3 KiB
JSON
{
|
|
"sample_id": "-iRBcNs9oI8/8",
|
|
"selection": "manually selected example with valid audio and vision query maps",
|
|
"duration_s": 8.093000411987305,
|
|
"word_count": 21,
|
|
"words": [
|
|
"I've",
|
|
"drawn",
|
|
"this",
|
|
"man",
|
|
"here,",
|
|
"and",
|
|
"if",
|
|
"you",
|
|
"can",
|
|
"tell,",
|
|
"he",
|
|
"has",
|
|
"a",
|
|
"lot",
|
|
"of",
|
|
"big",
|
|
"muscles,",
|
|
"he",
|
|
"looks",
|
|
"very",
|
|
"strong"
|
|
],
|
|
"audio_query_shape": [
|
|
21,
|
|
803
|
|
],
|
|
"audio_query_nnz": 495,
|
|
"audio_query_mapped_words": 21,
|
|
"vision_query_shape": [
|
|
21,
|
|
241
|
|
],
|
|
"vision_query_nnz": 160,
|
|
"vision_query_mapped_words": 21,
|
|
"audio_query_feature_validity_channel": "log_energy (audio feature dimension 66)",
|
|
"vision_query_feature_validity_channel": "OpenFace AU12 intensity (vision feature dimension 10)",
|
|
"alignment_semantics": "per-word normalized overlap between stored CTC word intervals and native source intervals; no learned content similarity",
|
|
"coordinate_axes_shown": false,
|
|
"color_scale": "shared square-root intensity transform for visibility; original overlap weights retained",
|
|
"word_label_positions": "actual transcript labels placed at their CTC word-interval centers on the source-time axis",
|
|
"image": "alignment_query_example.png"
|
|
}
|