{"id":54,"date":"2020-01-28T06:52:20","date_gmt":"2020-01-28T06:52:20","guid":{"rendered":"https:\/\/sail.usc.edu\/~mica\/wordpress\/?page_id=54"},"modified":"2021-10-07T04:26:27","modified_gmt":"2021-10-07T04:26:27","slug":"publications","status":"publish","type":"page","link":"https:\/\/sail.usc.edu:\/ccmi\/publications\/","title":{"rendered":"Publications"},"content":{"rendered":"\n<div class=\"teachpress_pub_list\"><form name=\"tppublistform\" method=\"get\"><a name=\"tppubs\" id=\"tppubs\"><\/a><div class=\"teachpress_cloud\"><span style=\"font-size:11px;\"><a rel=\"nofollow\" href=\"https:\/\/sail.usc.edu:\/ccmi\/publications\/?tgid=29&amp;yr=&amp;type=&amp;usr=&amp;auth=#tppubs\" title=\"2 Publications\" class=\"\">active speaker localization<\/a><\/span> <span style=\"font-size:11px;\"><a rel=\"nofollow\" href=\"https:\/\/sail.usc.edu:\/ccmi\/publications\/?tgid=3&amp;yr=&amp;type=&amp;usr=&amp;auth=#tppubs\" title=\"2 Publications\" class=\"\">advertisements<\/a><\/span> <span style=\"font-size:11px;\"><a rel=\"nofollow\" href=\"https:\/\/sail.usc.edu:\/ccmi\/publications\/?tgid=30&amp;yr=&amp;type=&amp;usr=&amp;auth=#tppubs\" title=\"1 Publication\" class=\"\">audio-visual event detection<\/a><\/span> <span style=\"font-size:11px;\"><a rel=\"nofollow\" href=\"https:\/\/sail.usc.edu:\/ccmi\/publications\/?tgid=13&amp;yr=&amp;type=&amp;usr=&amp;auth=#tppubs\" title=\"1 Publication\" class=\"\">Auschwitz<\/a><\/span> <span style=\"font-size:11px;\"><a rel=\"nofollow\" href=\"https:\/\/sail.usc.edu:\/ccmi\/publications\/?tgid=1&amp;yr=&amp;type=&amp;usr=&amp;auth=#tppubs\" title=\"2 Publications\" class=\"\">autoencoders<\/a><\/span> <span style=\"font-size:23px;\"><a rel=\"nofollow\" href=\"https:\/\/sail.usc.edu:\/ccmi\/publications\/?tgid=22&amp;yr=&amp;type=&amp;usr=&amp;auth=#tppubs\" title=\"4 Publications\" class=\"\">computational media understanding<\/a><\/span> <span style=\"font-size:11px;\"><a rel=\"nofollow\" href=\"https:\/\/sail.usc.edu:\/ccmi\/publications\/?tgid=15&amp;yr=&amp;type=&amp;usr=&amp;auth=#tppubs\" title=\"1 Publication\" class=\"\">computational narrative modeling<\/a><\/span> <span style=\"font-size:17px;\"><a rel=\"nofollow\" href=\"https:\/\/sail.usc.edu:\/ccmi\/publications\/?tgid=6&amp;yr=&amp;type=&amp;usr=&amp;auth=#tppubs\" title=\"3 Publications\" class=\"\">content analysis<\/a><\/span> <span style=\"font-size:11px;\"><a rel=\"nofollow\" href=\"https:\/\/sail.usc.edu:\/ccmi\/publications\/?tgid=4&amp;yr=&amp;type=&amp;usr=&amp;auth=#tppubs\" title=\"2 Publications\" class=\"\">coreference resolution<\/a><\/span> <span style=\"font-size:11px;\"><a rel=\"nofollow\" href=\"https:\/\/sail.usc.edu:\/ccmi\/publications\/?tgid=26&amp;yr=&amp;type=&amp;usr=&amp;auth=#tppubs\" title=\"2 Publications\" class=\"\">cross-modal learning<\/a><\/span> <span style=\"font-size:11px;\"><a rel=\"nofollow\" href=\"https:\/\/sail.usc.edu:\/ccmi\/publications\/?tgid=21&amp;yr=&amp;type=&amp;usr=&amp;auth=#tppubs\" title=\"1 Publication\" class=\"\">face clustering<\/a><\/span> <span style=\"font-size:11px;\"><a rel=\"nofollow\" href=\"https:\/\/sail.usc.edu:\/ccmi\/publications\/?tgid=20&amp;yr=&amp;type=&amp;usr=&amp;auth=#tppubs\" title=\"1 Publication\" class=\"\">face diarization<\/a><\/span> <span style=\"font-size:11px;\"><a rel=\"nofollow\" href=\"https:\/\/sail.usc.edu:\/ccmi\/publications\/?tgid=5&amp;yr=&amp;type=&amp;usr=&amp;auth=#tppubs\" title=\"1 Publication\" class=\"\">gender representation<\/a><\/span> <span style=\"font-size:11px;\"><a rel=\"nofollow\" href=\"https:\/\/sail.usc.edu:\/ccmi\/publications\/?tgid=25&amp;yr=&amp;type=&amp;usr=&amp;auth=#tppubs\" title=\"1 Publication\" class=\"\">gendered analysis<\/a><\/span> <span style=\"font-size:11px;\"><a rel=\"nofollow\" href=\"https:\/\/sail.usc.edu:\/ccmi\/publications\/?tgid=10&amp;yr=&amp;type=&amp;usr=&amp;auth=#tppubs\" title=\"1 Publication\" class=\"\">Media<\/a><\/span> <span style=\"font-size:11px;\"><a rel=\"nofollow\" href=\"https:\/\/sail.usc.edu:\/ccmi\/publications\/?tgid=8&amp;yr=&amp;type=&amp;usr=&amp;auth=#tppubs\" title=\"1 Publication\" class=\"\">movie<\/a><\/span> <span style=\"font-size:35px;\"><a rel=\"nofollow\" href=\"https:\/\/sail.usc.edu:\/ccmi\/publications\/?tgid=24&amp;yr=&amp;type=&amp;usr=&amp;auth=#tppubs\" title=\"6 Publications\" class=\"\">multimedia understanding<\/a><\/span> <span style=\"font-size:35px;\"><a rel=\"nofollow\" href=\"https:\/\/sail.usc.edu:\/ccmi\/publications\/?tgid=7&amp;yr=&amp;type=&amp;usr=&amp;auth=#tppubs\" title=\"6 Publications\" class=\"\">multimodal<\/a><\/span> <span style=\"font-size:11px;\"><a rel=\"nofollow\" href=\"https:\/\/sail.usc.edu:\/ccmi\/publications\/?tgid=28&amp;yr=&amp;type=&amp;usr=&amp;auth=#tppubs\" title=\"1 Publication\" class=\"\">multiple instance learning<\/a><\/span> <span style=\"font-size:11px;\"><a rel=\"nofollow\" href=\"https:\/\/sail.usc.edu:\/ccmi\/publications\/?tgid=18&amp;yr=&amp;type=&amp;usr=&amp;auth=#tppubs\" title=\"1 Publication\" class=\"\">multiview correlation<\/a><\/span> <span style=\"font-size:11px;\"><a rel=\"nofollow\" href=\"https:\/\/sail.usc.edu:\/ccmi\/publications\/?tgid=32&amp;yr=&amp;type=&amp;usr=&amp;auth=#tppubs\" title=\"2 Publications\" class=\"\">music representations<\/a><\/span> <span style=\"font-size:11px;\"><a rel=\"nofollow\" href=\"https:\/\/sail.usc.edu:\/ccmi\/publications\/?tgid=9&amp;yr=&amp;type=&amp;usr=&amp;auth=#tppubs\" title=\"1 Publication\" class=\"\">Professions<\/a><\/span> <span style=\"font-size:17px;\"><a rel=\"nofollow\" href=\"https:\/\/sail.usc.edu:\/ccmi\/publications\/?tgid=17&amp;yr=&amp;type=&amp;usr=&amp;auth=#tppubs\" title=\"3 Publications\" class=\"\">self-supervision<\/a><\/span> <span style=\"font-size:11px;\"><a rel=\"nofollow\" href=\"https:\/\/sail.usc.edu:\/ccmi\/publications\/?tgid=23&amp;yr=&amp;type=&amp;usr=&amp;auth=#tppubs\" title=\"1 Publication\" class=\"\">semantic role labeling<\/a><\/span> <span style=\"font-size:11px;\"><a rel=\"nofollow\" href=\"https:\/\/sail.usc.edu:\/ccmi\/publications\/?tgid=14&amp;yr=&amp;type=&amp;usr=&amp;auth=#tppubs\" title=\"1 Publication\" class=\"\">survivor testimonies<\/a><\/span> <span style=\"font-size:11px;\"><a rel=\"nofollow\" href=\"https:\/\/sail.usc.edu:\/ccmi\/publications\/?tgid=12&amp;yr=&amp;type=&amp;usr=&amp;auth=#tppubs\" title=\"1 Publication\" class=\"\">taxonomy curation<\/a><\/span> <span style=\"font-size:11px;\"><a rel=\"nofollow\" href=\"https:\/\/sail.usc.edu:\/ccmi\/publications\/?tgid=19&amp;yr=&amp;type=&amp;usr=&amp;auth=#tppubs\" title=\"1 Publication\" class=\"\">triplet loss<\/a><\/span> <span style=\"font-size:11px;\"><a rel=\"nofollow\" href=\"https:\/\/sail.usc.edu:\/ccmi\/publications\/?tgid=16&amp;yr=&amp;type=&amp;usr=&amp;auth=#tppubs\" title=\"1 Publication\" class=\"\">video character labeling<\/a><\/span> <span style=\"font-size:11px;\"><a rel=\"nofollow\" href=\"https:\/\/sail.usc.edu:\/ccmi\/publications\/?tgid=11&amp;yr=&amp;type=&amp;usr=&amp;auth=#tppubs\" title=\"1 Publication\" class=\"\">visual scene recognition<\/a><\/span> <span style=\"font-size:11px;\"><a rel=\"nofollow\" href=\"https:\/\/sail.usc.edu:\/ccmi\/publications\/?tgid=27&amp;yr=&amp;type=&amp;usr=&amp;auth=#tppubs\" title=\"1 Publication\" class=\"\">weakly supervised learning<\/a><\/span> <\/div><div class=\"teachpress_filter\"><select class=\"default\" name=\"yr\" id=\"yr\" tabindex=\"2\" onchange=\"teachpress_jumpMenu('parent',this, 'https:\/\/sail.usc.edu:\/ccmi\/publications\/?')\">\n                   <option value=\"tgid=&amp;yr=&amp;type=&amp;usr=&amp;auth=#tppubs\">All years<\/option>\n                   <option value = \"tgid=&amp;yr=2023&amp;type=&amp;usr=&amp;auth=#tppubs\" >2023<\/option><option value = \"tgid=&amp;yr=2022&amp;type=&amp;usr=&amp;auth=#tppubs\" >2022<\/option><option value = \"tgid=&amp;yr=2021&amp;type=&amp;usr=&amp;auth=#tppubs\" >2021<\/option><option value = \"tgid=&amp;yr=2020&amp;type=&amp;usr=&amp;auth=#tppubs\" >2020<\/option><option value = \"tgid=&amp;yr=2019&amp;type=&amp;usr=&amp;auth=#tppubs\" >2019<\/option><option value = \"tgid=&amp;yr=2018&amp;type=&amp;usr=&amp;auth=#tppubs\" >2018<\/option><option value = \"tgid=&amp;yr=2017&amp;type=&amp;usr=&amp;auth=#tppubs\" >2017<\/option><option value = \"tgid=&amp;yr=2016&amp;type=&amp;usr=&amp;auth=#tppubs\" >2016<\/option><option value = \"tgid=&amp;yr=2015&amp;type=&amp;usr=&amp;auth=#tppubs\" >2015<\/option>\n                <\/select><select class=\"default\" name=\"type\" id=\"type\" tabindex=\"3\" onchange=\"teachpress_jumpMenu('parent',this, 'https:\/\/sail.usc.edu:\/ccmi\/publications\/?')\">\n                   <option value=\"tgid=&amp;yr=&amp;type=&amp;usr=&amp;auth=#tppubs\">All types<\/option>\n                   <option value = \"tgid=&amp;yr=&amp;type=article&amp;usr=&amp;auth=#tppubs\" >Journal Articles<\/option><option value = \"tgid=&amp;yr=&amp;type=conference&amp;usr=&amp;auth=#tppubs\" >Conferences<\/option><option value = \"tgid=&amp;yr=&amp;type=inproceedings&amp;usr=&amp;auth=#tppubs\" >Inproceedings<\/option>\n                <\/select><select class=\"default\" name=\"auth\" id=\"auth\" tabindex=\"5\" onchange=\"teachpress_jumpMenu('parent',this, 'https:\/\/sail.usc.edu:\/ccmi\/publications\/?')\">\n                   <option value=\"tgid=&amp;yr=&amp;type=&amp;usr=&amp;auth=#tppubs\">All authors<\/option>\n                   <option value = \"tgid=&amp;yr=&amp;type=&amp;usr=&amp;auth=16#tppubs\" > Adam, Hartwig<\/option><option value = \"tgid=&amp;yr=&amp;type=&amp;usr=&amp;auth=56#tppubs\" > Avramidis, Kleanthis<\/option><option value = \"tgid=&amp;yr=&amp;type=&amp;usr=&amp;auth=28#tppubs\" > Baruah, Sabyasachee<\/option><option value = \"tgid=&amp;yr=&amp;type=&amp;usr=&amp;auth=49#tppubs\" > Bose, Digbalay<\/option><option value = \"tgid=&amp;yr=&amp;type=&amp;usr=&amp;auth=29#tppubs\" > Chakravarthula, Sandeep Nallan<\/option><option value = \"tgid=&amp;yr=&amp;type=&amp;usr=&amp;auth=54#tppubs\" > Cole-McLaughlin, Kree<\/option><option value = \"tgid=&amp;yr=&amp;type=&amp;usr=&amp;auth=53#tppubs\" > Cui, Yin<\/option><option value = \"tgid=&amp;yr=&amp;type=&amp;usr=&amp;auth=59#tppubs\" > Feng, Tiantian<\/option><option value = \"tgid=&amp;yr=&amp;type=&amp;usr=&amp;auth=33#tppubs\" > Georgiou, Panayiotis<\/option><option value = \"tgid=&amp;yr=&amp;type=&amp;usr=&amp;auth=31#tppubs\" > Goyal, Ankit<\/option><option value = \"tgid=&amp;yr=&amp;type=&amp;usr=&amp;auth=10#tppubs\" > Greer, Timothy<\/option><option value = \"tgid=&amp;yr=&amp;type=&amp;usr=&amp;auth=13#tppubs\" > Guha, Tanaya<\/option><option value = \"tgid=&amp;yr=&amp;type=&amp;usr=&amp;auth=19#tppubs\" > Gupta, Rahul<\/option><option value = \"tgid=&amp;yr=&amp;type=&amp;usr=&amp;auth=2#tppubs\" > Hebbar, Rajat<\/option><option value = \"tgid=&amp;yr=&amp;type=&amp;usr=&amp;auth=51#tppubs\" > Hempel, Tim<\/option><option value = \"tgid=&amp;yr=&amp;type=&amp;usr=&amp;auth=37#tppubs\" > Huang, Che-Wei<\/option><option value = \"tgid=&amp;yr=&amp;type=&amp;usr=&amp;auth=9#tppubs\" > Knox, Dillon<\/option><option value = \"tgid=&amp;yr=&amp;type=&amp;usr=&amp;auth=15#tppubs\" > Kumar, Naveen<\/option><option value = \"tgid=&amp;yr=&amp;type=&amp;usr=&amp;auth=12#tppubs\" > Kuo, Emily<\/option><option value = \"tgid=&amp;yr=&amp;type=&amp;usr=&amp;auth=11#tppubs\" > Ma, Benjamin<\/option><option value = \"tgid=&amp;yr=&amp;type=&amp;usr=&amp;auth=17#tppubs\" > Madni, Asad M<\/option><option value = \"tgid=&amp;yr=&amp;type=&amp;usr=&amp;auth=27#tppubs\" > Malandrakis, Nikolaos<\/option><option value = \"tgid=&amp;yr=&amp;type=&amp;usr=&amp;auth=24#tppubs\" > Martinez, Victor<\/option><option value = \"tgid=&amp;yr=&amp;type=&amp;usr=&amp;auth=14#tppubs\" > Martinez, Victor R<\/option><option value = \"tgid=&amp;yr=&amp;type=&amp;usr=&amp;auth=8#tppubs\" > Narayanan, Shrikanth<\/option><option value = \"tgid=&amp;yr=&amp;type=&amp;usr=&amp;auth=3#tppubs\" > Narayanan, Shrikanth S<\/option><option value = \"tgid=&amp;yr=&amp;type=&amp;usr=&amp;auth=32#tppubs\" > Nasir, Md<\/option><option value = \"tgid=&amp;yr=&amp;type=&amp;usr=&amp;auth=4#tppubs\" > Peri, Raghuveer<\/option><option value = \"tgid=&amp;yr=&amp;type=&amp;usr=&amp;auth=18#tppubs\" > Ramakrishna, Anil Kumar<\/option><option value = \"tgid=&amp;yr=&amp;type=&amp;usr=&amp;auth=22#tppubs\" > Ramakrishna, Anil<\/option><option value = \"tgid=&amp;yr=&amp;type=&amp;usr=&amp;auth=7#tppubs\" > Rivera, Fernando<\/option><option value = \"tgid=&amp;yr=&amp;type=&amp;usr=&amp;auth=20#tppubs\" > Sharma, Rahul<\/option><option value = \"tgid=&amp;yr=&amp;type=&amp;usr=&amp;auth=47#tppubs\" > Shi, Xuan<\/option><option value = \"tgid=&amp;yr=&amp;type=&amp;usr=&amp;auth=21#tppubs\" > Singla, Karan<\/option><option value = \"tgid=&amp;yr=&amp;type=&amp;usr=&amp;auth=36#tppubs\" > Smith, Stacy L<\/option><option value = \"tgid=&amp;yr=&amp;type=&amp;usr=&amp;auth=1#tppubs\" > Somandepalli, Krishna<\/option><option value = \"tgid=&amp;yr=&amp;type=&amp;usr=&amp;auth=57#tppubs\" > Stewart, Shanti<\/option><option value = \"tgid=&amp;yr=&amp;type=&amp;usr=&amp;auth=34#tppubs\" > Tadimari, Adarsh<\/option><option value = \"tgid=&amp;yr=&amp;type=&amp;usr=&amp;auth=25#tppubs\" > Tehranian-Uhls, Yalda<\/option><option value = \"tgid=&amp;yr=&amp;type=&amp;usr=&amp;auth=50#tppubs\" > T\u00f3th, G\u00e1bor Mih\u00e1ly<\/option><option value = \"tgid=&amp;yr=&amp;type=&amp;usr=&amp;auth=5#tppubs\" > Travadi, Ruchir<\/option><option value = \"tgid=&amp;yr=&amp;type=&amp;usr=&amp;auth=6#tppubs\" > Tuplin, Tracy<\/option><option value = \"tgid=&amp;yr=&amp;type=&amp;usr=&amp;auth=23#tppubs\" > Uhls, Yalda T<\/option><option value = \"tgid=&amp;yr=&amp;type=&amp;usr=&amp;auth=58#tppubs\" > Vijai, Veena<\/option><option value = \"tgid=&amp;yr=&amp;type=&amp;usr=&amp;auth=55#tppubs\" > Wang, Huisheng<\/option><option value = \"tgid=&amp;yr=&amp;type=&amp;usr=&amp;auth=60#tppubs\" > Xu, Anfeng<\/option><option value = \"tgid=&amp;yr=&amp;type=&amp;usr=&amp;auth=52#tppubs\" > Zhang, Haoyang<\/option><option value = \"tgid=&amp;yr=&amp;type=&amp;usr=&amp;auth=38#tppubs\" > Zhu, Yan<\/option>\n                <\/select><select class=\"default\" name=\"usr\" id=\"usr\" tabindex=\"6\" onchange=\"teachpress_jumpMenu('parent',this, 'https:\/\/sail.usc.edu:\/ccmi\/publications\/?')\">\n                   <option value=\"tgid=&amp;yr=&amp;type=&amp;usr=&amp;auth=#tppubs\">All users<\/option>\n                   \n                <\/select><\/div><\/form><div class=\"teachpress_publication_list\"><h3 class=\"tp_h3\" id=\"tp_h3_2023\">2023<\/h3><div class=\"tp_publication tp_publication_conference\"><div class=\"tp_pub_info\"><p class=\"tp_pub_author\"> Hebbar, Rajat;  Bose, Digbalay;  Narayanan, Shrikanth<\/p><p class=\"tp_pub_title\">SEAR: Semantically-grounded Audio Representations <span class=\"tp_pub_type conference\">Conference<\/span> <\/p><p class=\"tp_pub_additional\"><span class=\"tp_pub_additional_publisher\">ACM Multimedia , <\/span><span class=\"tp_pub_additional_year\">2023<\/span>.<\/p><p class=\"tp_pub_menu\"><span class=\"tp_bibtex_link\"><a id=\"tp_bibtex_sh_42\" class=\"tp_show\" onclick=\"teachpress_pub_showhide('42','tp_bibtex')\" title=\"Show BibTeX entry\" style=\"cursor:pointer;\">BibTeX<\/a><\/span> | <span class=\"tp_pub_tags_label\">Tags: <\/span><a rel=\"nofollow\" href=\"https:\/\/sail.usc.edu:\/ccmi\/publications\/?tgid=22#tppubs\" title=\"Show all publications which have a relationship to this tag\">computational media understanding<\/a>, <a rel=\"nofollow\" href=\"https:\/\/sail.usc.edu:\/ccmi\/publications\/?tgid=7#tppubs\" title=\"Show all publications which have a relationship to this tag\">multimodal<\/a>, <a rel=\"nofollow\" href=\"https:\/\/sail.usc.edu:\/ccmi\/publications\/?tgid=17#tppubs\" title=\"Show all publications which have a relationship to this tag\">self-supervision<\/a><\/p><div class=\"tp_bibtex\" id=\"tp_bibtex_42\" style=\"display:none;\"><div class=\"tp_bibtex_entry\"><pre>@conference{nokey,<br \/>\r\ntitle = {SEAR: Semantically-grounded Audio Representations},<br \/>\r\nauthor = {Rajat Hebbar and Digbalay Bose and Shrikanth Narayanan},<br \/>\r\nyear  = {2023},<br \/>\r\ndate = {2023-10-29},<br \/>\r\npublisher = {ACM Multimedia },<br \/>\r\nkeywords = {computational media understanding, multimodal, self-supervision},<br \/>\r\npubstate = {published},<br \/>\r\ntppubtype = {conference}<br \/>\r\n}<br \/>\r\n<\/pre><\/div><p class=\"tp_close_menu\"><a class=\"tp_close\" onclick=\"teachpress_pub_showhide('42','tp_bibtex')\">Close<\/a><\/p><\/div><\/div><\/div><div class=\"tp_publication tp_publication_conference\"><div class=\"tp_pub_info\"><p class=\"tp_pub_author\"> Bose, Digbalay;  Hebbar, Rajat;  Feng, Tiantian;  Somandepalli, Krishna;  Xu, Anfeng;  Narayanan, Shrikanth<\/p><p class=\"tp_pub_title\">MM-AU: Towards Multimodal Understanding of Advertisement Videos <span class=\"tp_pub_type conference\">Conference<\/span> <\/p><p class=\"tp_pub_additional\"><span class=\"tp_pub_additional_publisher\">ACM Multimedia , <\/span><span class=\"tp_pub_additional_year\">2023<\/span>.<\/p><p class=\"tp_pub_menu\"><span class=\"tp_bibtex_link\"><a id=\"tp_bibtex_sh_41\" class=\"tp_show\" onclick=\"teachpress_pub_showhide('41','tp_bibtex')\" title=\"Show BibTeX entry\" style=\"cursor:pointer;\">BibTeX<\/a><\/span> | <span class=\"tp_pub_tags_label\">Tags: <\/span><a rel=\"nofollow\" href=\"https:\/\/sail.usc.edu:\/ccmi\/publications\/?tgid=3#tppubs\" title=\"Show all publications which have a relationship to this tag\">advertisements<\/a>, <a rel=\"nofollow\" href=\"https:\/\/sail.usc.edu:\/ccmi\/publications\/?tgid=22#tppubs\" title=\"Show all publications which have a relationship to this tag\">computational media understanding<\/a>, <a rel=\"nofollow\" href=\"https:\/\/sail.usc.edu:\/ccmi\/publications\/?tgid=6#tppubs\" title=\"Show all publications which have a relationship to this tag\">content analysis<\/a>, <a rel=\"nofollow\" href=\"https:\/\/sail.usc.edu:\/ccmi\/publications\/?tgid=24#tppubs\" title=\"Show all publications which have a relationship to this tag\">multimedia understanding<\/a>, <a rel=\"nofollow\" href=\"https:\/\/sail.usc.edu:\/ccmi\/publications\/?tgid=7#tppubs\" title=\"Show all publications which have a relationship to this tag\">multimodal<\/a><\/p><div class=\"tp_bibtex\" id=\"tp_bibtex_41\" style=\"display:none;\"><div class=\"tp_bibtex_entry\"><pre>@conference{nokey,<br \/>\r\ntitle = {MM-AU: Towards Multimodal Understanding of Advertisement Videos},<br \/>\r\nauthor = {Digbalay Bose and Rajat Hebbar and Tiantian Feng and Krishna Somandepalli and Anfeng Xu and Shrikanth Narayanan },<br \/>\r\nyear  = {2023},<br \/>\r\ndate = {2023-10-29},<br \/>\r\nurldate = {2023-10-29},<br \/>\r\npublisher = {ACM Multimedia },<br \/>\r\nkeywords = {advertisements, computational media understanding, content analysis, multimedia understanding, multimodal},<br \/>\r\npubstate = {published},<br \/>\r\ntppubtype = {conference}<br \/>\r\n}<br \/>\r\n<\/pre><\/div><p class=\"tp_close_menu\"><a class=\"tp_close\" onclick=\"teachpress_pub_showhide('41','tp_bibtex')\">Close<\/a><\/p><\/div><\/div><\/div><div class=\"tp_publication tp_publication_article\"><div class=\"tp_pub_info\"><p class=\"tp_pub_author\"> Sharma, Rahul;  Narayanan, Shrikanth<\/p><p class=\"tp_pub_title\"><a class=\"tp_title_link\" onclick=\"teachpress_pub_showhide('40','tp_links')\" style=\"cursor:pointer;\">Audio-Visual Activity Guided Cross-Modal Identity Association for Active Speaker Detection<\/a> <span class=\"tp_pub_type article\">Journal Article<\/span> <\/p><p class=\"tp_pub_additional\"><span class=\"tp_pub_additional_in\">In: <\/span><span class=\"tp_pub_additional_journal\">IEEE Open Journal of Signal Processing , <\/span><span class=\"tp_pub_additional_pages\">pp. 225-232, <\/span><span class=\"tp_pub_additional_year\">2023<\/span>.<\/p><p class=\"tp_pub_menu\"><span class=\"tp_abstract_link\"><a id=\"tp_abstract_sh_40\" class=\"tp_show\" onclick=\"teachpress_pub_showhide('40','tp_abstract')\" title=\"Show abstract\" style=\"cursor:pointer;\">Abstract<\/a><\/span> | <span class=\"tp_resource_link\"><a id=\"tp_links_sh_40\" class=\"tp_show\" onclick=\"teachpress_pub_showhide('40','tp_links')\" title=\"Show links and resources\" style=\"cursor:pointer;\">Links<\/a><\/span> | <span class=\"tp_bibtex_link\"><a id=\"tp_bibtex_sh_40\" class=\"tp_show\" onclick=\"teachpress_pub_showhide('40','tp_bibtex')\" title=\"Show BibTeX entry\" style=\"cursor:pointer;\">BibTeX<\/a><\/span> | <span class=\"tp_pub_tags_label\">Tags: <\/span><a rel=\"nofollow\" href=\"https:\/\/sail.usc.edu:\/ccmi\/publications\/?tgid=29#tppubs\" title=\"Show all publications which have a relationship to this tag\">active speaker localization<\/a>, <a rel=\"nofollow\" href=\"https:\/\/sail.usc.edu:\/ccmi\/publications\/?tgid=22#tppubs\" title=\"Show all publications which have a relationship to this tag\">computational media understanding<\/a>, <a rel=\"nofollow\" href=\"https:\/\/sail.usc.edu:\/ccmi\/publications\/?tgid=26#tppubs\" title=\"Show all publications which have a relationship to this tag\">cross-modal learning<\/a>, <a rel=\"nofollow\" href=\"https:\/\/sail.usc.edu:\/ccmi\/publications\/?tgid=24#tppubs\" title=\"Show all publications which have a relationship to this tag\">multimedia understanding<\/a><\/p><div class=\"tp_bibtex\" id=\"tp_bibtex_40\" style=\"display:none;\"><div class=\"tp_bibtex_entry\"><pre>@article{nokey,<br \/>\r\ntitle = {Audio-Visual Activity Guided Cross-Modal Identity Association for Active Speaker Detection},<br \/>\r\nauthor = {Rahul Sharma and Shrikanth Narayanan},<br \/>\r\ndoi = {10.1109\/OJSP.2023.3267269},<br \/>\r\nyear  = {2023},<br \/>\r\ndate = {2023-04-14},<br \/>\r\nurldate = {2023-04-14},<br \/>\r\njournal = {IEEE Open Journal of Signal Processing },<br \/>\r\npages = {225-232},<br \/>\r\nabstract = {Active speaker detection in videos addresses associating a source face, visible in the video frames, with the underlying speech in the audio modality. The two primary sources of information to derive such a speech-face relationship are i) visual activity and its interaction with the speech signal and ii) co-occurrences of speakers' identities across modalities in the form of face and speech. The two approaches have their limitations: the audio-visual activity models get confused with other frequently occurring vocal activities, such as laughing and chewing, while the speakers' identity-based methods are limited to videos having enough disambiguating information to establish a speech-face association. Since the two approaches are independent, we investigate their complementary nature in this work. We propose a novel unsupervised framework to guide the speakers' cross-modal identity association with the audio-visual activity for active speaker detection. Through experiments on entertainment media videos from two benchmark datasets\u2013the AVA active speaker (movies) and Visual Person Clustering Dataset (TV shows)\u2013we show that a simple late fusion of the two approaches enhances the active speaker detection performance.},<br \/>\r\nkeywords = {active speaker localization, computational media understanding, cross-modal learning, multimedia understanding},<br \/>\r\npubstate = {published},<br \/>\r\ntppubtype = {article}<br \/>\r\n}<br \/>\r\n<\/pre><\/div><p class=\"tp_close_menu\"><a class=\"tp_close\" onclick=\"teachpress_pub_showhide('40','tp_bibtex')\">Close<\/a><\/p><\/div><div class=\"tp_abstract\" id=\"tp_abstract_40\" style=\"display:none;\"><div class=\"tp_abstract_entry\">Active speaker detection in videos addresses associating a source face, visible in the video frames, with the underlying speech in the audio modality. The two primary sources of information to derive such a speech-face relationship are i) visual activity and its interaction with the speech signal and ii) co-occurrences of speakers&#039; identities across modalities in the form of face and speech. The two approaches have their limitations: the audio-visual activity models get confused with other frequently occurring vocal activities, such as laughing and chewing, while the speakers&#039; identity-based methods are limited to videos having enough disambiguating information to establish a speech-face association. Since the two approaches are independent, we investigate their complementary nature in this work. We propose a novel unsupervised framework to guide the speakers&#039; cross-modal identity association with the audio-visual activity for active speaker detection. Through experiments on entertainment media videos from two benchmark datasets\u2013the AVA active speaker (movies) and Visual Person Clustering Dataset (TV shows)\u2013we show that a simple late fusion of the two approaches enhances the active speaker detection performance.<\/div><p class=\"tp_close_menu\"><a class=\"tp_close\" onclick=\"teachpress_pub_showhide('40','tp_abstract')\">Close<\/a><\/p><\/div><div class=\"tp_links\" id=\"tp_links_40\" style=\"display:none;\"><div class=\"tp_links_entry\"><ul class=\"tp_pub_list\"><li><i class=\"ai ai-doi\"><\/i><a class=\"tp_pub_list\" href=\"https:\/\/dx.doi.org\/10.1109\/OJSP.2023.3267269\" title=\"Follow DOI:10.1109\/OJSP.2023.3267269\" target=\"_blank\">doi:10.1109\/OJSP.2023.3267269<\/a><\/li><\/ul><\/div><p class=\"tp_close_menu\"><a class=\"tp_close\" onclick=\"teachpress_pub_showhide('40','tp_links')\">Close<\/a><\/p><\/div><\/div><\/div><div class=\"tp_publication tp_publication_conference\"><div class=\"tp_pub_info\"><p class=\"tp_pub_author\"> Bose, Digbalay;  Hebbar, Rajat;  Somandepalli, Krishna;  Narayanan, Shrikanth<\/p><p class=\"tp_pub_title\"><a class=\"tp_title_link\" onclick=\"teachpress_pub_showhide('36','tp_links')\" style=\"cursor:pointer;\">Contextually-rich human affect perception using multimodal scene information<\/a> <span class=\"tp_pub_type conference\">Conference<\/span> <\/p><p class=\"tp_pub_additional\"><span class=\"tp_pub_additional_publisher\">IEEE International Conference on Acoustics, Speech, and Signal Processing (ICASSP) , <\/span><span class=\"tp_pub_additional_year\">2023<\/span>.<\/p><p class=\"tp_pub_menu\"><span class=\"tp_abstract_link\"><a id=\"tp_abstract_sh_36\" class=\"tp_show\" onclick=\"teachpress_pub_showhide('36','tp_abstract')\" title=\"Show abstract\" style=\"cursor:pointer;\">Abstract<\/a><\/span> | <span class=\"tp_resource_link\"><a id=\"tp_links_sh_36\" class=\"tp_show\" onclick=\"teachpress_pub_showhide('36','tp_links')\" title=\"Show links and resources\" style=\"cursor:pointer;\">Links<\/a><\/span> | <span class=\"tp_bibtex_link\"><a id=\"tp_bibtex_sh_36\" class=\"tp_show\" onclick=\"teachpress_pub_showhide('36','tp_bibtex')\" title=\"Show BibTeX entry\" style=\"cursor:pointer;\">BibTeX<\/a><\/span> | <span class=\"tp_pub_tags_label\">Tags: <\/span><a rel=\"nofollow\" href=\"https:\/\/sail.usc.edu:\/ccmi\/publications\/?tgid=31#tppubs\" title=\"Show all publications which have a relationship to this tag\">emotion recognition<\/a>, <a rel=\"nofollow\" href=\"https:\/\/sail.usc.edu:\/ccmi\/publications\/?tgid=24#tppubs\" title=\"Show all publications which have a relationship to this tag\">multimedia understanding<\/a>, <a rel=\"nofollow\" href=\"https:\/\/sail.usc.edu:\/ccmi\/publications\/?tgid=7#tppubs\" title=\"Show all publications which have a relationship to this tag\">multimodal<\/a><\/p><div class=\"tp_bibtex\" id=\"tp_bibtex_36\" style=\"display:none;\"><div class=\"tp_bibtex_entry\"><pre>@conference{bose-etal-2023-emotion-recognition,<br \/>\r\ntitle = {Contextually-rich human affect perception using multimodal scene information},<br \/>\r\nauthor = {Digbalay Bose and Rajat Hebbar and Krishna Somandepalli and Shrikanth Narayanan },<br \/>\r\ndoi = {https:\/\/arxiv.org\/abs\/2303.06904},<br \/>\r\nyear  = {2023},<br \/>\r\ndate = {2023-03-13},<br \/>\r\nurldate = {2023-03-13},<br \/>\r\npublisher = {IEEE International Conference on Acoustics, Speech, and Signal Processing (ICASSP) },<br \/>\r\nabstract = {The process of human affect understanding involves the ability to infer person specific emotional states from various sources including images, speech, and language. Affect perception from images has predominantly focused on expressions extracted from salient face crops. However, emotions perceived by humans rely on multiple contextual cues including social settings, foreground interactions, and ambient visual scenes. In this work, we leverage pretrained vision-language (VLN) models to extract descriptions of foreground context from images. Further, we propose a multimodal context fusion (MCF) module to combine foreground cues with the visual scene and person-based contextual information for emotion prediction. We show the effectiveness of our proposed modular design on two datasets associated with natural scenes and TV shows.},<br \/>\r\nkeywords = {emotion recognition, multimedia understanding, multimodal},<br \/>\r\npubstate = {published},<br \/>\r\ntppubtype = {conference}<br \/>\r\n}<br \/>\r\n<\/pre><\/div><p class=\"tp_close_menu\"><a class=\"tp_close\" onclick=\"teachpress_pub_showhide('36','tp_bibtex')\">Close<\/a><\/p><\/div><div class=\"tp_abstract\" id=\"tp_abstract_36\" style=\"display:none;\"><div class=\"tp_abstract_entry\">The process of human affect understanding involves the ability to infer person specific emotional states from various sources including images, speech, and language. Affect perception from images has predominantly focused on expressions extracted from salient face crops. However, emotions perceived by humans rely on multiple contextual cues including social settings, foreground interactions, and ambient visual scenes. In this work, we leverage pretrained vision-language (VLN) models to extract descriptions of foreground context from images. Further, we propose a multimodal context fusion (MCF) module to combine foreground cues with the visual scene and person-based contextual information for emotion prediction. We show the effectiveness of our proposed modular design on two datasets associated with natural scenes and TV shows.<\/div><p class=\"tp_close_menu\"><a class=\"tp_close\" onclick=\"teachpress_pub_showhide('36','tp_abstract')\">Close<\/a><\/p><\/div><div class=\"tp_links\" id=\"tp_links_36\" style=\"display:none;\"><div class=\"tp_links_entry\"><ul class=\"tp_pub_list\"><li><i class=\"ai ai-doi\"><\/i><a class=\"tp_pub_list\" href=\"https:\/\/dx.doi.org\/https:\/\/arxiv.org\/abs\/2303.06904\" title=\"Follow DOI:https:\/\/arxiv.org\/abs\/2303.06904\" target=\"_blank\">doi:https:\/\/arxiv.org\/abs\/2303.06904<\/a><\/li><\/ul><\/div><p class=\"tp_close_menu\"><a class=\"tp_close\" onclick=\"teachpress_pub_showhide('36','tp_links')\">Close<\/a><\/p><\/div><\/div><\/div><div class=\"tp_publication tp_publication_conference\"><div class=\"tp_pub_info\"><p class=\"tp_pub_author\"> Avramidis, Kleanthis;  Stewart, Shanti;  Narayanan, Shrikanth<\/p><p class=\"tp_pub_title\"><a class=\"tp_title_link\" onclick=\"teachpress_pub_showhide('37','tp_links')\" style=\"cursor:pointer;\">On the Role of Visual Context in Enriching Music Representations<\/a> <span class=\"tp_pub_type conference\">Conference<\/span> <\/p><p class=\"tp_pub_additional\"><span class=\"tp_pub_additional_publisher\">IEEE International Conference on Acoustics, Speech, and Signal Processing (ICASSP) , <\/span><span class=\"tp_pub_additional_year\">2023<\/span>.<\/p><p class=\"tp_pub_menu\"><span class=\"tp_abstract_link\"><a id=\"tp_abstract_sh_37\" class=\"tp_show\" onclick=\"teachpress_pub_showhide('37','tp_abstract')\" title=\"Show abstract\" style=\"cursor:pointer;\">Abstract<\/a><\/span> | <span class=\"tp_resource_link\"><a id=\"tp_links_sh_37\" class=\"tp_show\" onclick=\"teachpress_pub_showhide('37','tp_links')\" title=\"Show links and resources\" style=\"cursor:pointer;\">Links<\/a><\/span> | <span class=\"tp_bibtex_link\"><a id=\"tp_bibtex_sh_37\" class=\"tp_show\" onclick=\"teachpress_pub_showhide('37','tp_bibtex')\" title=\"Show BibTeX entry\" style=\"cursor:pointer;\">BibTeX<\/a><\/span> | <span class=\"tp_pub_tags_label\">Tags: <\/span><a rel=\"nofollow\" href=\"https:\/\/sail.usc.edu:\/ccmi\/publications\/?tgid=24#tppubs\" title=\"Show all publications which have a relationship to this tag\">multimedia understanding<\/a>, <a rel=\"nofollow\" href=\"https:\/\/sail.usc.edu:\/ccmi\/publications\/?tgid=7#tppubs\" title=\"Show all publications which have a relationship to this tag\">multimodal<\/a>, <a rel=\"nofollow\" href=\"https:\/\/sail.usc.edu:\/ccmi\/publications\/?tgid=32#tppubs\" title=\"Show all publications which have a relationship to this tag\">music representations<\/a><\/p><div class=\"tp_bibtex\" id=\"tp_bibtex_37\" style=\"display:none;\"><div class=\"tp_bibtex_entry\"><pre>@conference{avramidi-etal-vcmr,<br \/>\r\ntitle = {On the Role of Visual Context in Enriching Music Representations},<br \/>\r\nauthor = {Kleanthis Avramidis and Shanti Stewart and Shrikanth Narayanan},<br \/>\r\ndoi = {https:\/\/arxiv.org\/abs\/2210.15828},<br \/>\r\nyear  = {2023},<br \/>\r\ndate = {2023-02-15},<br \/>\r\nurldate = {2023-02-15},<br \/>\r\npublisher = {IEEE International Conference on Acoustics, Speech, and Signal Processing (ICASSP) },<br \/>\r\nabstract = {Human perception and experience of music is highly context-dependent. Contextual variability contributes to differences in how we interpret and interact with music, challenging the design of robust models for information retrieval. Incorporating multimodal context from diverse sources provides a promising approach toward modeling this variability. Music presented in media such as movies and music videos provide rich multimodal context that modulates underlying human experiences. However, such context modeling is underexplored, as it requires large amounts of multimodal data along with relevant annotations. Self-supervised learning can help address these challenges by automatically extracting rich, high-level correspondences between different modalities, hence alleviating the need for fine-grained annotations at scale. In this study, we propose VCMR -- Video-Conditioned Music Representations, a contrastive learning framework that learns music representations from audio and the accompanying music videos. The contextual visual information enhances representations of music audio, as evaluated on the downstream task of music tagging. Experimental results show that the proposed framework can contribute additive robustness to audio representations and indicates to what extent musical elements are affected or determined by visual context.},<br \/>\r\nkeywords = {multimedia understanding, multimodal, music representations},<br \/>\r\npubstate = {published},<br \/>\r\ntppubtype = {conference}<br \/>\r\n}<br \/>\r\n<\/pre><\/div><p class=\"tp_close_menu\"><a class=\"tp_close\" onclick=\"teachpress_pub_showhide('37','tp_bibtex')\">Close<\/a><\/p><\/div><div class=\"tp_abstract\" id=\"tp_abstract_37\" style=\"display:none;\"><div class=\"tp_abstract_entry\">Human perception and experience of music is highly context-dependent. Contextual variability contributes to differences in how we interpret and interact with music, challenging the design of robust models for information retrieval. Incorporating multimodal context from diverse sources provides a promising approach toward modeling this variability. Music presented in media such as movies and music videos provide rich multimodal context that modulates underlying human experiences. However, such context modeling is underexplored, as it requires large amounts of multimodal data along with relevant annotations. Self-supervised learning can help address these challenges by automatically extracting rich, high-level correspondences between different modalities, hence alleviating the need for fine-grained annotations at scale. In this study, we propose VCMR -- Video-Conditioned Music Representations, a contrastive learning framework that learns music representations from audio and the accompanying music videos. The contextual visual information enhances representations of music audio, as evaluated on the downstream task of music tagging. Experimental results show that the proposed framework can contribute additive robustness to audio representations and indicates to what extent musical elements are affected or determined by visual context.<\/div><p class=\"tp_close_menu\"><a class=\"tp_close\" onclick=\"teachpress_pub_showhide('37','tp_abstract')\">Close<\/a><\/p><\/div><div class=\"tp_links\" id=\"tp_links_37\" style=\"display:none;\"><div class=\"tp_links_entry\"><ul class=\"tp_pub_list\"><li><i class=\"ai ai-doi\"><\/i><a class=\"tp_pub_list\" href=\"https:\/\/dx.doi.org\/https:\/\/arxiv.org\/abs\/2210.15828\" title=\"Follow DOI:https:\/\/arxiv.org\/abs\/2210.15828\" target=\"_blank\">doi:https:\/\/arxiv.org\/abs\/2210.15828<\/a><\/li><\/ul><\/div><p class=\"tp_close_menu\"><a class=\"tp_close\" onclick=\"teachpress_pub_showhide('37','tp_links')\">Close<\/a><\/p><\/div><\/div><\/div><div class=\"tp_publication tp_publication_conference\"><div class=\"tp_pub_info\"><p class=\"tp_pub_author\"> Hebbar, Rajat;  Bose, Digbalay;  Somandepalli, Krishna;  Vijai, Veena;  Narayanan, Shrikanth<\/p><p class=\"tp_pub_title\"><a class=\"tp_title_link\" onclick=\"teachpress_pub_showhide('35','tp_links')\" style=\"cursor:pointer;\">A dataset for Audio-Visual Sound Event Detection in Movies<\/a> <span class=\"tp_pub_type conference\">Conference<\/span> <\/p><p class=\"tp_pub_additional\"><span class=\"tp_pub_additional_publisher\">IEEE International Conference on Acoustics, Speech, and Signal Processing (ICASSP) , <\/span><span class=\"tp_pub_additional_year\">2023<\/span>.<\/p><p class=\"tp_pub_menu\"><span class=\"tp_abstract_link\"><a id=\"tp_abstract_sh_35\" class=\"tp_show\" onclick=\"teachpress_pub_showhide('35','tp_abstract')\" title=\"Show abstract\" style=\"cursor:pointer;\">Abstract<\/a><\/span> | <span class=\"tp_resource_link\"><a id=\"tp_links_sh_35\" class=\"tp_show\" onclick=\"teachpress_pub_showhide('35','tp_links')\" title=\"Show links and resources\" style=\"cursor:pointer;\">Links<\/a><\/span> | <span class=\"tp_bibtex_link\"><a id=\"tp_bibtex_sh_35\" class=\"tp_show\" onclick=\"teachpress_pub_showhide('35','tp_bibtex')\" title=\"Show BibTeX entry\" style=\"cursor:pointer;\">BibTeX<\/a><\/span> | <span class=\"tp_pub_tags_label\">Tags: <\/span><a rel=\"nofollow\" href=\"https:\/\/sail.usc.edu:\/ccmi\/publications\/?tgid=30#tppubs\" title=\"Show all publications which have a relationship to this tag\">audio-visual event detection<\/a>, <a rel=\"nofollow\" href=\"https:\/\/sail.usc.edu:\/ccmi\/publications\/?tgid=7#tppubs\" title=\"Show all publications which have a relationship to this tag\">multimodal<\/a><\/p><div class=\"tp_bibtex\" id=\"tp_bibtex_35\" style=\"display:none;\"><div class=\"tp_bibtex_entry\"><pre>@conference{hebbar-2023-audio-events,<br \/>\r\ntitle = {A dataset for Audio-Visual Sound Event Detection in Movies},<br \/>\r\nauthor = {Rajat Hebbar and Digbalay Bose and Krishna Somandepalli and Veena Vijai and Shrikanth Narayanan},<br \/>\r\ndoi = {https:\/\/arxiv.org\/abs\/2302.07315},<br \/>\r\nyear  = {2023},<br \/>\r\ndate = {2023-02-14},<br \/>\r\nurldate = {2023-02-14},<br \/>\r\npublisher = {IEEE International Conference on Acoustics, Speech, and Signal Processing (ICASSP) },<br \/>\r\nabstract = {Audio event detection is a widely studied audio processing task, with applications ranging from self-driving cars to healthcare. In-the-wild datasets such as Audioset have propelled research in this field. However, many efforts typically involve manual annotation and verification, which is expensive to perform at scale. Movies depict various real-life and fictional scenarios which makes them a rich resource for mining a wide-range of audio events. In this work, we present a dataset of audio events called Subtitle-Aligned Movie Sounds (SAM-S). We use publicly-available closed-caption transcripts to automatically mine over 110K audio events from 430 movies. We identify three dimensions to categorize audio events: sound, source, quality, and present the steps involved to produce a final taxonomy of 245 sounds. We discuss the choices involved in generating the taxonomy, and also highlight the human-centered nature of sounds in our dataset. We establish a baseline performance for audio-only sound classification of 34.76% mean average precision and show that incorporating visual information can further improve the performance by about 5%.},<br \/>\r\nkeywords = {audio-visual event detection, multimodal},<br \/>\r\npubstate = {published},<br \/>\r\ntppubtype = {conference}<br \/>\r\n}<br \/>\r\n<\/pre><\/div><p class=\"tp_close_menu\"><a class=\"tp_close\" onclick=\"teachpress_pub_showhide('35','tp_bibtex')\">Close<\/a><\/p><\/div><div class=\"tp_abstract\" id=\"tp_abstract_35\" style=\"display:none;\"><div class=\"tp_abstract_entry\">Audio event detection is a widely studied audio processing task, with applications ranging from self-driving cars to healthcare. In-the-wild datasets such as Audioset have propelled research in this field. However, many efforts typically involve manual annotation and verification, which is expensive to perform at scale. Movies depict various real-life and fictional scenarios which makes them a rich resource for mining a wide-range of audio events. In this work, we present a dataset of audio events called Subtitle-Aligned Movie Sounds (SAM-S). We use publicly-available closed-caption transcripts to automatically mine over 110K audio events from 430 movies. We identify three dimensions to categorize audio events: sound, source, quality, and present the steps involved to produce a final taxonomy of 245 sounds. We discuss the choices involved in generating the taxonomy, and also highlight the human-centered nature of sounds in our dataset. We establish a baseline performance for audio-only sound classification of 34.76% mean average precision and show that incorporating visual information can further improve the performance by about 5%.<\/div><p class=\"tp_close_menu\"><a class=\"tp_close\" onclick=\"teachpress_pub_showhide('35','tp_abstract')\">Close<\/a><\/p><\/div><div class=\"tp_links\" id=\"tp_links_35\" style=\"display:none;\"><div class=\"tp_links_entry\"><ul class=\"tp_pub_list\"><li><i class=\"ai ai-doi\"><\/i><a class=\"tp_pub_list\" href=\"https:\/\/dx.doi.org\/https:\/\/arxiv.org\/abs\/2302.07315\" title=\"Follow DOI:https:\/\/arxiv.org\/abs\/2302.07315\" target=\"_blank\">doi:https:\/\/arxiv.org\/abs\/2302.07315<\/a><\/li><\/ul><\/div><p class=\"tp_close_menu\"><a class=\"tp_close\" onclick=\"teachpress_pub_showhide('35','tp_links')\">Close<\/a><\/p><\/div><\/div><\/div><div class=\"tp_publication tp_publication_inproceedings\"><div class=\"tp_pub_info\"><p class=\"tp_pub_author\"> Baruah, Sabyasachee;  Narayanan, Shrikanth<\/p><p class=\"tp_pub_title\">Character Coreference Resolution in Movie Screenplays <span class=\"tp_pub_type inproceedings\">Inproceedings<\/span> <\/p><p class=\"tp_pub_additional\"><span class=\"tp_pub_additional_in\">In: <\/span><span class=\"tp_pub_additional_booktitle\">Findings of the Association for Computational Linguistics: ACL 2023, <\/span><span class=\"tp_pub_additional_pages\">pp. 10300\u201310313, <\/span><span class=\"tp_pub_additional_year\">2023<\/span>.<\/p><p class=\"tp_pub_menu\"><span class=\"tp_abstract_link\"><a id=\"tp_abstract_sh_39\" class=\"tp_show\" onclick=\"teachpress_pub_showhide('39','tp_abstract')\" title=\"Show abstract\" style=\"cursor:pointer;\">Abstract<\/a><\/span> | <span class=\"tp_bibtex_link\"><a id=\"tp_bibtex_sh_39\" class=\"tp_show\" onclick=\"teachpress_pub_showhide('39','tp_bibtex')\" title=\"Show BibTeX entry\" style=\"cursor:pointer;\">BibTeX<\/a><\/span> | <span class=\"tp_pub_tags_label\">Tags: <\/span><a rel=\"nofollow\" href=\"https:\/\/sail.usc.edu:\/ccmi\/publications\/?tgid=6#tppubs\" title=\"Show all publications which have a relationship to this tag\">content analysis<\/a>, <a rel=\"nofollow\" href=\"https:\/\/sail.usc.edu:\/ccmi\/publications\/?tgid=4#tppubs\" title=\"Show all publications which have a relationship to this tag\">coreference resolution<\/a>, <a rel=\"nofollow\" href=\"https:\/\/sail.usc.edu:\/ccmi\/publications\/?tgid=24#tppubs\" title=\"Show all publications which have a relationship to this tag\">multimedia understanding<\/a><\/p><div class=\"tp_bibtex\" id=\"tp_bibtex_39\" style=\"display:none;\"><div class=\"tp_bibtex_entry\"><pre>@inproceedings{baruah2023character,<br \/>\r\ntitle = {Character Coreference Resolution in Movie Screenplays},<br \/>\r\nauthor = {Sabyasachee Baruah and Shrikanth Narayanan},<br \/>\r\nyear  = {2023},<br \/>\r\ndate = {2023-01-01},<br \/>\r\nurldate = {2023-01-01},<br \/>\r\nbooktitle = {Findings of the Association for Computational Linguistics: ACL 2023},<br \/>\r\npages = {10300--10313},<br \/>\r\nabstract = {Movie screenplays have a distinct narrative structure. It segments the story into scenes containing interleaving descriptions of actions, locations, and character dialogues. A typical screenplay spans several scenes and can include long-range dependencies between characters and events. A holistic document-level understanding of the screenplay requires several natural language processing capabilities, such as parsing, character identification, coreference resolution, action recognition, summarization, and attribute discovery. In this work, we develop scalable and robust methods to extract the structural information and character coreference clusters from full-length movie screenplays. We curate two datasets for screenplay parsing and character coreference\u2014 MovieParse and MovieCoref, respectively. We build a robust screenplay parser to handle inconsistencies in screenplay formatting and<br \/>\r\nleverage the parsed output to link co-referring character mentions. Our coreference models can scale to long screenplay documents without drastically increasing their memory footprints.},<br \/>\r\nkeywords = {content analysis, coreference resolution, multimedia understanding},<br \/>\r\npubstate = {published},<br \/>\r\ntppubtype = {inproceedings}<br \/>\r\n}<br \/>\r\n<\/pre><\/div><p class=\"tp_close_menu\"><a class=\"tp_close\" onclick=\"teachpress_pub_showhide('39','tp_bibtex')\">Close<\/a><\/p><\/div><div class=\"tp_abstract\" id=\"tp_abstract_39\" style=\"display:none;\"><div class=\"tp_abstract_entry\">Movie screenplays have a distinct narrative structure. It segments the story into scenes containing interleaving descriptions of actions, locations, and character dialogues. A typical screenplay spans several scenes and can include long-range dependencies between characters and events. A holistic document-level understanding of the screenplay requires several natural language processing capabilities, such as parsing, character identification, coreference resolution, action recognition, summarization, and attribute discovery. In this work, we develop scalable and robust methods to extract the structural information and character coreference clusters from full-length movie screenplays. We curate two datasets for screenplay parsing and character coreference\u2014 MovieParse and MovieCoref, respectively. We build a robust screenplay parser to handle inconsistencies in screenplay formatting and<br \/>\r\nleverage the parsed output to link co-referring character mentions. Our coreference models can scale to long screenplay documents without drastically increasing their memory footprints.<\/div><p class=\"tp_close_menu\"><a class=\"tp_close\" onclick=\"teachpress_pub_showhide('39','tp_abstract')\">Close<\/a><\/p><\/div><\/div><\/div><div class=\"tp_publication tp_publication_article\"><div class=\"tp_pub_info\"><p class=\"tp_pub_author\"> Greer, Timothy;  Shi, Xuan;  Ma, Benjamin;  Narayanan, Shrikanth<\/p><p class=\"tp_pub_title\">Creating musical features using multi-faceted, multi-task encoders based on transformers <span class=\"tp_pub_type article\">Journal Article<\/span> <\/p><p class=\"tp_pub_additional\"><span class=\"tp_pub_additional_in\">In: <\/span><span class=\"tp_pub_additional_journal\">Scientific Reports, <\/span><span class=\"tp_pub_additional_volume\">13 <\/span><span class=\"tp_pub_additional_number\">(1), <\/span><span class=\"tp_pub_additional_pages\">pp. 10713, <\/span><span class=\"tp_pub_additional_year\">2023<\/span>.<\/p><p class=\"tp_pub_menu\"><span class=\"tp_abstract_link\"><a id=\"tp_abstract_sh_38\" class=\"tp_show\" onclick=\"teachpress_pub_showhide('38','tp_abstract')\" title=\"Show abstract\" style=\"cursor:pointer;\">Abstract<\/a><\/span> | <span class=\"tp_bibtex_link\"><a id=\"tp_bibtex_sh_38\" class=\"tp_show\" onclick=\"teachpress_pub_showhide('38','tp_bibtex')\" title=\"Show BibTeX entry\" style=\"cursor:pointer;\">BibTeX<\/a><\/span> | <span class=\"tp_pub_tags_label\">Tags: <\/span><a rel=\"nofollow\" href=\"https:\/\/sail.usc.edu:\/ccmi\/publications\/?tgid=1#tppubs\" title=\"Show all publications which have a relationship to this tag\">autoencoders<\/a>, <a rel=\"nofollow\" href=\"https:\/\/sail.usc.edu:\/ccmi\/publications\/?tgid=32#tppubs\" title=\"Show all publications which have a relationship to this tag\">music representations<\/a>, <a rel=\"nofollow\" href=\"https:\/\/sail.usc.edu:\/ccmi\/publications\/?tgid=17#tppubs\" title=\"Show all publications which have a relationship to this tag\">self-supervision<\/a><\/p><div class=\"tp_bibtex\" id=\"tp_bibtex_38\" style=\"display:none;\"><div class=\"tp_bibtex_entry\"><pre>@article{greer2023creating,<br \/>\r\ntitle = {Creating musical features using multi-faceted, multi-task encoders based on transformers},<br \/>\r\nauthor = {Timothy Greer and Xuan Shi and Benjamin Ma and Shrikanth Narayanan},<br \/>\r\nyear  = {2023},<br \/>\r\ndate = {2023-01-01},<br \/>\r\nurldate = {2023-01-01},<br \/>\r\njournal = {Scientific Reports},<br \/>\r\nvolume = {13},<br \/>\r\nnumber = {1},<br \/>\r\npages = {10713},<br \/>\r\npublisher = {Nature Publishing Group UK London},<br \/>\r\nabstract = {Computational machine intelligence approaches have enabled a variety of music-centric technologies in support of creating, sharing and interacting with music content. A strong performance on specific downstream application tasks, such as music genre detection and music emotion recognition, is paramount to ensuring broad capabilities for computational music understanding and Music Information Retrieval. Traditional approaches have relied on supervised learning to train models to support these music-related tasks. However, such approaches require copious annotated data and still may only provide insight into one view of music\u2014namely, that related to the specific task at hand. We present a new model for generating audio-musical features that support music understanding, leveraging self-supervision and cross-domain learning. After pre-training using masked reconstruction of musical input features using self-attention bidirectional transformers, output representations are fine-tuned using several downstream music understanding tasks. Results show that the features generated by our multi-faceted, multi-task, music transformer model, which we call M3BERT, tend to outperform other audio and music embeddings on several diverse music-related tasks, indicating the potential of self-supervised and semi-supervised learning approaches toward a more generalized and robust computational approach to modeling music. Our work can offer a starting point for many music-related modeling tasks, with potential applications in learning deep representations and enabling robust technology applications.},<br \/>\r\nkeywords = {autoencoders, music representations, self-supervision},<br \/>\r\npubstate = {published},<br \/>\r\ntppubtype = {article}<br \/>\r\n}<br \/>\r\n<\/pre><\/div><p class=\"tp_close_menu\"><a class=\"tp_close\" onclick=\"teachpress_pub_showhide('38','tp_bibtex')\">Close<\/a><\/p><\/div><div class=\"tp_abstract\" id=\"tp_abstract_38\" style=\"display:none;\"><div class=\"tp_abstract_entry\">Computational machine intelligence approaches have enabled a variety of music-centric technologies in support of creating, sharing and interacting with music content. A strong performance on specific downstream application tasks, such as music genre detection and music emotion recognition, is paramount to ensuring broad capabilities for computational music understanding and Music Information Retrieval. Traditional approaches have relied on supervised learning to train models to support these music-related tasks. However, such approaches require copious annotated data and still may only provide insight into one view of music\u2014namely, that related to the specific task at hand. We present a new model for generating audio-musical features that support music understanding, leveraging self-supervision and cross-domain learning. After pre-training using masked reconstruction of musical input features using self-attention bidirectional transformers, output representations are fine-tuned using several downstream music understanding tasks. Results show that the features generated by our multi-faceted, multi-task, music transformer model, which we call M3BERT, tend to outperform other audio and music embeddings on several diverse music-related tasks, indicating the potential of self-supervised and semi-supervised learning approaches toward a more generalized and robust computational approach to modeling music. Our work can offer a starting point for many music-related modeling tasks, with potential applications in learning deep representations and enabling robust technology applications.<\/div><p class=\"tp_close_menu\"><a class=\"tp_close\" onclick=\"teachpress_pub_showhide('38','tp_abstract')\">Close<\/a><\/p><\/div><\/div><\/div><h3 class=\"tp_h3\" id=\"tp_h3_2022\">2022<\/h3><div class=\"tp_publication tp_publication_article\"><div class=\"tp_pub_info\"><p class=\"tp_pub_author\"> Martinez, Victor;  Somandepalli, Krishna;  Narayanan, Shrikanth<\/p><p class=\"tp_pub_title\">Boys don\u2019t cry (or kiss or dance): A computational linguistic lens into gendered actions in film <span class=\"tp_pub_type article\">Journal Article<\/span> <\/p><p class=\"tp_pub_additional\"><span class=\"tp_pub_additional_in\">In: <\/span><span class=\"tp_pub_additional_journal\">PLoS One, <\/span><span class=\"tp_pub_additional_year\">2022<\/span>.<\/p><p class=\"tp_pub_menu\"><span class=\"tp_abstract_link\"><a id=\"tp_abstract_sh_33\" class=\"tp_show\" onclick=\"teachpress_pub_showhide('33','tp_abstract')\" title=\"Show abstract\" style=\"cursor:pointer;\">Abstract<\/a><\/span> | <span class=\"tp_bibtex_link\"><a id=\"tp_bibtex_sh_33\" class=\"tp_show\" onclick=\"teachpress_pub_showhide('33','tp_bibtex')\" title=\"Show BibTeX entry\" style=\"cursor:pointer;\">BibTeX<\/a><\/span> | <span class=\"tp_pub_tags_label\">Tags: <\/span><a rel=\"nofollow\" href=\"https:\/\/sail.usc.edu:\/ccmi\/publications\/?tgid=25#tppubs\" title=\"Show all publications which have a relationship to this tag\">gendered analysis<\/a>, <a rel=\"nofollow\" href=\"https:\/\/sail.usc.edu:\/ccmi\/publications\/?tgid=24#tppubs\" title=\"Show all publications which have a relationship to this tag\">multimedia understanding<\/a>, <a rel=\"nofollow\" href=\"https:\/\/sail.usc.edu:\/ccmi\/publications\/?tgid=23#tppubs\" title=\"Show all publications which have a relationship to this tag\">semantic role labeling<\/a><\/p><div class=\"tp_bibtex\" id=\"tp_bibtex_33\" style=\"display:none;\"><div class=\"tp_bibtex_entry\"><pre>@article{Martinez2022,<br \/>\r\ntitle = {Boys don\u2019t cry (or kiss or dance): A computational linguistic lens into gendered actions in film},<br \/>\r\nauthor = {Victor Martinez and Krishna Somandepalli and Shrikanth Narayanan},<br \/>\r\nyear  = {2022},<br \/>\r\ndate = {2022-12-20},<br \/>\r\nurldate = {2022-12-20},<br \/>\r\njournal = {PLoS One},<br \/>\r\nabstract = { },<br \/>\r\nkeywords = {gendered analysis, multimedia understanding, semantic role labeling},<br \/>\r\npubstate = {published},<br \/>\r\ntppubtype = {article}<br \/>\r\n}<br \/>\r\n<\/pre><\/div><p class=\"tp_close_menu\"><a class=\"tp_close\" onclick=\"teachpress_pub_showhide('33','tp_bibtex')\">Close<\/a><\/p><\/div><div class=\"tp_abstract\" id=\"tp_abstract_33\" style=\"display:none;\"><div class=\"tp_abstract_entry\"> <\/div><p class=\"tp_close_menu\"><a class=\"tp_close\" onclick=\"teachpress_pub_showhide('33','tp_abstract')\">Close<\/a><\/p><\/div><\/div><\/div><div class=\"tp_publication tp_publication_article\"><div class=\"tp_pub_info\"><p class=\"tp_pub_author\"> Sharma, Rahul;  Somandepalli, Krishna;  Narayanan, Shrikanth<\/p><p class=\"tp_pub_title\"><a class=\"tp_title_link\" onclick=\"teachpress_pub_showhide('34','tp_links')\" style=\"cursor:pointer;\">Cross modal video representations for weakly supervised active speaker localization<\/a> <span class=\"tp_pub_type article\">Journal Article<\/span> <\/p><p class=\"tp_pub_additional\"><span class=\"tp_pub_additional_in\">In: <\/span><span class=\"tp_pub_additional_journal\">IEEE Transactions on Multimedia, <\/span><span class=\"tp_pub_additional_volume\">Early Access <\/span>, <span class=\"tp_pub_additional_pages\">pp. 1-12, <\/span><span class=\"tp_pub_additional_year\">2022<\/span>.<\/p><p class=\"tp_pub_menu\"><span class=\"tp_abstract_link\"><a id=\"tp_abstract_sh_34\" class=\"tp_show\" onclick=\"teachpress_pub_showhide('34','tp_abstract')\" title=\"Show abstract\" style=\"cursor:pointer;\">Abstract<\/a><\/span> | <span class=\"tp_resource_link\"><a id=\"tp_links_sh_34\" class=\"tp_show\" onclick=\"teachpress_pub_showhide('34','tp_links')\" title=\"Show links and resources\" style=\"cursor:pointer;\">Links<\/a><\/span> | <span class=\"tp_bibtex_link\"><a id=\"tp_bibtex_sh_34\" class=\"tp_show\" onclick=\"teachpress_pub_showhide('34','tp_bibtex')\" title=\"Show BibTeX entry\" style=\"cursor:pointer;\">BibTeX<\/a><\/span> | <span class=\"tp_pub_tags_label\">Tags: <\/span><a rel=\"nofollow\" href=\"https:\/\/sail.usc.edu:\/ccmi\/publications\/?tgid=29#tppubs\" title=\"Show all publications which have a relationship to this tag\">active speaker localization<\/a>, <a rel=\"nofollow\" href=\"https:\/\/sail.usc.edu:\/ccmi\/publications\/?tgid=26#tppubs\" title=\"Show all publications which have a relationship to this tag\">cross-modal learning<\/a>, <a rel=\"nofollow\" href=\"https:\/\/sail.usc.edu:\/ccmi\/publications\/?tgid=28#tppubs\" title=\"Show all publications which have a relationship to this tag\">multiple instance learning<\/a>, <a rel=\"nofollow\" href=\"https:\/\/sail.usc.edu:\/ccmi\/publications\/?tgid=27#tppubs\" title=\"Show all publications which have a relationship to this tag\">weakly supervised learning<\/a><\/p><div class=\"tp_bibtex\" id=\"tp_bibtex_34\" style=\"display:none;\"><div class=\"tp_bibtex_entry\"><pre>@article{Sharma2022,<br \/>\r\ntitle = {Cross modal video representations for weakly supervised active speaker localization},<br \/>\r\nauthor = {Rahul Sharma and Krishna Somandepalli and Shrikanth Narayanan},<br \/>\r\nurl = {https:\/\/ieeexplore.ieee.org\/document\/9991097},<br \/>\r\ndoi = {10.1109\/TMM.2022.3229975},<br \/>\r\nyear  = {2022},<br \/>\r\ndate = {2022-12-16},<br \/>\r\nurldate = {2022-12-16},<br \/>\r\njournal = {IEEE Transactions on Multimedia},<br \/>\r\nvolume = {Early Access},<br \/>\r\npages = {1-12},<br \/>\r\nabstract = {An objective understanding of media depictions, such as inclusive portrayals of how much someone is heard and seen on screen such as in film and television, requires the machines to discern automatically who, when, how, and where someone is talking, and not. Speaker activity can be automatically discerned from the rich multimodal information present in the media content. This is however a challenging problem due to the vast variety and contextual variability in media content, and the lack of labeled data. In this work, we present a cross-modal neural network for learning visual representations, which have implicit information pertaining to the spatial location of a speaker in the visual frames. Avoiding the need for manual annotations for active speakers in visual frames, acquiring of which is very expensive, we present a weakly supervised system for the task of localizing active speakers in movie content. We use the learned cross-modal visual representations, and provide weak supervision from movie subtitles acting as a proxy for voice activity, thus requiring no manual annotations. Furthermore, we propose an audio-assisted post-processing formulation for the task of active speaker detection. We evaluate the performance of the proposed system on three benchmark datasets: i) AVA active speaker dataset, ii) Visual person clustering dataset, and iii) Columbia datset, and demonstrate the effectiveness of the cross-modal embeddings for localizing active speakers in comparison to fully supervised systems.},<br \/>\r\nkeywords = {active speaker localization, cross-modal learning, multiple instance learning, weakly supervised learning},<br \/>\r\npubstate = {published},<br \/>\r\ntppubtype = {article}<br \/>\r\n}<br \/>\r\n<\/pre><\/div><p class=\"tp_close_menu\"><a class=\"tp_close\" onclick=\"teachpress_pub_showhide('34','tp_bibtex')\">Close<\/a><\/p><\/div><div class=\"tp_abstract\" id=\"tp_abstract_34\" style=\"display:none;\"><div class=\"tp_abstract_entry\">An objective understanding of media depictions, such as inclusive portrayals of how much someone is heard and seen on screen such as in film and television, requires the machines to discern automatically who, when, how, and where someone is talking, and not. Speaker activity can be automatically discerned from the rich multimodal information present in the media content. This is however a challenging problem due to the vast variety and contextual variability in media content, and the lack of labeled data. In this work, we present a cross-modal neural network for learning visual representations, which have implicit information pertaining to the spatial location of a speaker in the visual frames. Avoiding the need for manual annotations for active speakers in visual frames, acquiring of which is very expensive, we present a weakly supervised system for the task of localizing active speakers in movie content. We use the learned cross-modal visual representations, and provide weak supervision from movie subtitles acting as a proxy for voice activity, thus requiring no manual annotations. Furthermore, we propose an audio-assisted post-processing formulation for the task of active speaker detection. We evaluate the performance of the proposed system on three benchmark datasets: i) AVA active speaker dataset, ii) Visual person clustering dataset, and iii) Columbia datset, and demonstrate the effectiveness of the cross-modal embeddings for localizing active speakers in comparison to fully supervised systems.<\/div><p class=\"tp_close_menu\"><a class=\"tp_close\" onclick=\"teachpress_pub_showhide('34','tp_abstract')\">Close<\/a><\/p><\/div><div class=\"tp_links\" id=\"tp_links_34\" style=\"display:none;\"><div class=\"tp_links_entry\"><ul class=\"tp_pub_list\"><li><i class=\"fas fa-globe\"><\/i><a class=\"tp_pub_list\" href=\"https:\/\/ieeexplore.ieee.org\/document\/9991097\" title=\"https:\/\/ieeexplore.ieee.org\/document\/9991097\" target=\"_blank\">https:\/\/ieeexplore.ieee.org\/document\/9991097<\/a><\/li><li><i class=\"ai ai-doi\"><\/i><a class=\"tp_pub_list\" href=\"https:\/\/dx.doi.org\/10.1109\/TMM.2022.3229975\" title=\"Follow DOI:10.1109\/TMM.2022.3229975\" target=\"_blank\">doi:10.1109\/TMM.2022.3229975<\/a><\/li><\/ul><\/div><p class=\"tp_close_menu\"><a class=\"tp_close\" onclick=\"teachpress_pub_showhide('34','tp_links')\">Close<\/a><\/p><\/div><\/div><\/div><div class=\"tp_publication tp_publication_conference\"><div class=\"tp_pub_info\"><p class=\"tp_pub_author\"> Bose, Digbalay;  Hebbar, Rajat;  Somandepalli, Krishna;  Zhang, Haoyang;  Cui, Yin;  Cole-McLaughlin, Kree;  Wang, Huisheng;  Narayanan, Shrikanth<\/p><p class=\"tp_pub_title\"><a class=\"tp_title_link\" onclick=\"teachpress_pub_showhide('30','tp_links')\" style=\"cursor:pointer;\">MovieCLIP: Visual Scene Recognition in Movies<\/a> <span class=\"tp_pub_type conference\">Conference<\/span> <\/p><p class=\"tp_pub_additional\"><span class=\"tp_pub_additional_publisher\">IEEE\/CVF Winter Conference on Applications of Computer Vision (WACV 2023), <\/span><span class=\"tp_pub_additional_year\">2022<\/span>.<\/p><p class=\"tp_pub_menu\"><span class=\"tp_abstract_link\"><a id=\"tp_abstract_sh_30\" class=\"tp_show\" onclick=\"teachpress_pub_showhide('30','tp_abstract')\" title=\"Show abstract\" style=\"cursor:pointer;\">Abstract<\/a><\/span> | <span class=\"tp_resource_link\"><a id=\"tp_links_sh_30\" class=\"tp_show\" onclick=\"teachpress_pub_showhide('30','tp_links')\" title=\"Show links and resources\" style=\"cursor:pointer;\">Links<\/a><\/span> | <span class=\"tp_bibtex_link\"><a id=\"tp_bibtex_sh_30\" class=\"tp_show\" onclick=\"teachpress_pub_showhide('30','tp_bibtex')\" title=\"Show BibTeX entry\" style=\"cursor:pointer;\">BibTeX<\/a><\/span> | <span class=\"tp_pub_tags_label\">Tags: <\/span><a rel=\"nofollow\" href=\"https:\/\/sail.usc.edu:\/ccmi\/publications\/?tgid=12#tppubs\" title=\"Show all publications which have a relationship to this tag\">taxonomy curation<\/a>, <a rel=\"nofollow\" href=\"https:\/\/sail.usc.edu:\/ccmi\/publications\/?tgid=11#tppubs\" title=\"Show all publications which have a relationship to this tag\">visual scene recognition<\/a><\/p><div class=\"tp_bibtex\" id=\"tp_bibtex_30\" style=\"display:none;\"><div class=\"tp_bibtex_entry\"><pre>@conference{bose-etal-2022-visual-scene,<br \/>\r\ntitle = {MovieCLIP: Visual Scene Recognition in Movies},<br \/>\r\nauthor = {Digbalay Bose and Rajat Hebbar and Krishna Somandepalli and Haoyang Zhang and Yin Cui and Kree Cole-McLaughlin and Huisheng Wang and Shrikanth Narayanan},<br \/>\r\nurl = {https:\/\/arxiv.org\/abs\/2210.11065},<br \/>\r\nyear  = {2022},<br \/>\r\ndate = {2022-10-23},<br \/>\r\nurldate = {2022-10-23},<br \/>\r\npublisher = {IEEE\/CVF Winter Conference on Applications of Computer Vision (WACV 2023)},<br \/>\r\nabstract = {Longform media such as movies have complex narrative structures, with events spanning a rich variety of ambient visual scenes. Domain-specific challenges associated with visual scenes in movies include transitions, person coverage, and a wide array of real-life and fictional scenarios. Existing visual scene datasets in movies have limited taxonomies and don't consider the visual scene transition within movie clips. In this work, we address the problem of visual scene recognition in movies by first automatically curating a new and extensive movie-centric taxonomy of 179 scene labels derived from movie scripts and auxiliary web-based video datasets. Instead of manual annotations which can be expensive, we use CLIP to weakly label 1.12 million shots from 32K movie clips based on our proposed taxonomy. We provide baseline visual models trained on the weakly labeled dataset called MovieCLIP and evaluate them on an independent dataset verified by human raters. We show that leveraging features from models pretrained on MovieCLIP benefits downstream tasks such as multi-label scene and genre classification of web videos and movie trailers.},<br \/>\r\nkeywords = {taxonomy curation, visual scene recognition},<br \/>\r\npubstate = {published},<br \/>\r\ntppubtype = {conference}<br \/>\r\n}<br \/>\r\n<\/pre><\/div><p class=\"tp_close_menu\"><a class=\"tp_close\" onclick=\"teachpress_pub_showhide('30','tp_bibtex')\">Close<\/a><\/p><\/div><div class=\"tp_abstract\" id=\"tp_abstract_30\" style=\"display:none;\"><div class=\"tp_abstract_entry\">Longform media such as movies have complex narrative structures, with events spanning a rich variety of ambient visual scenes. Domain-specific challenges associated with visual scenes in movies include transitions, person coverage, and a wide array of real-life and fictional scenarios. Existing visual scene datasets in movies have limited taxonomies and don&#039;t consider the visual scene transition within movie clips. In this work, we address the problem of visual scene recognition in movies by first automatically curating a new and extensive movie-centric taxonomy of 179 scene labels derived from movie scripts and auxiliary web-based video datasets. Instead of manual annotations which can be expensive, we use CLIP to weakly label 1.12 million shots from 32K movie clips based on our proposed taxonomy. We provide baseline visual models trained on the weakly labeled dataset called MovieCLIP and evaluate them on an independent dataset verified by human raters. We show that leveraging features from models pretrained on MovieCLIP benefits downstream tasks such as multi-label scene and genre classification of web videos and movie trailers.<\/div><p class=\"tp_close_menu\"><a class=\"tp_close\" onclick=\"teachpress_pub_showhide('30','tp_abstract')\">Close<\/a><\/p><\/div><div class=\"tp_links\" id=\"tp_links_30\" style=\"display:none;\"><div class=\"tp_links_entry\"><ul class=\"tp_pub_list\"><li><i class=\"ai ai-arxiv\"><\/i><a class=\"tp_pub_list\" href=\"https:\/\/arxiv.org\/abs\/2210.11065\" title=\"https:\/\/arxiv.org\/abs\/2210.11065\" target=\"_blank\">https:\/\/arxiv.org\/abs\/2210.11065<\/a><\/li><\/ul><\/div><p class=\"tp_close_menu\"><a class=\"tp_close\" onclick=\"teachpress_pub_showhide('30','tp_links')\">Close<\/a><\/p><\/div><\/div><\/div><div class=\"tp_publication tp_publication_article\"><div class=\"tp_pub_info\"><p class=\"tp_pub_author\"> T\u00f3th, G\u00e1bor Mih\u00e1ly;  Hempel, Tim;  Somandepalli, Krishna;  Narayanan, Shrikanth<\/p><p class=\"tp_pub_title\"><a class=\"tp_title_link\" onclick=\"teachpress_pub_showhide('31','tp_links')\" style=\"cursor:pointer;\">Studying Large-Scale Behavioral Differences in Auschwitz-Birkenau with Simulation of Gendered Narratives<\/a> <span class=\"tp_pub_type article\">Journal Article<\/span> <\/p><p class=\"tp_pub_additional\"><span class=\"tp_pub_additional_in\">In: <\/span><span class=\"tp_pub_additional_journal\">Digital Humanities Quarterly, <\/span><span class=\"tp_pub_additional_volume\">16 <\/span><span class=\"tp_pub_additional_number\">(3), <\/span><span class=\"tp_pub_additional_year\">2022<\/span>.<\/p><p class=\"tp_pub_menu\"><span class=\"tp_abstract_link\"><a id=\"tp_abstract_sh_31\" class=\"tp_show\" onclick=\"teachpress_pub_showhide('31','tp_abstract')\" title=\"Show abstract\" style=\"cursor:pointer;\">Abstract<\/a><\/span> | <span class=\"tp_resource_link\"><a id=\"tp_links_sh_31\" class=\"tp_show\" onclick=\"teachpress_pub_showhide('31','tp_links')\" title=\"Show links and resources\" style=\"cursor:pointer;\">Links<\/a><\/span> | <span class=\"tp_bibtex_link\"><a id=\"tp_bibtex_sh_31\" class=\"tp_show\" onclick=\"teachpress_pub_showhide('31','tp_bibtex')\" title=\"Show BibTeX entry\" style=\"cursor:pointer;\">BibTeX<\/a><\/span> | <span class=\"tp_pub_tags_label\">Tags: <\/span><a rel=\"nofollow\" href=\"https:\/\/sail.usc.edu:\/ccmi\/publications\/?tgid=13#tppubs\" title=\"Show all publications which have a relationship to this tag\">Auschwitz<\/a>, <a rel=\"nofollow\" href=\"https:\/\/sail.usc.edu:\/ccmi\/publications\/?tgid=15#tppubs\" title=\"Show all publications which have a relationship to this tag\">computational narrative modeling<\/a>, <a rel=\"nofollow\" href=\"https:\/\/sail.usc.edu:\/ccmi\/publications\/?tgid=14#tppubs\" title=\"Show all publications which have a relationship to this tag\">survivor testimonies<\/a><\/p><div class=\"tp_bibtex\" id=\"tp_bibtex_31\" style=\"display:none;\"><div class=\"tp_bibtex_entry\"><pre>@article{T\u00f3th2022,<br \/>\r\ntitle = {Studying Large-Scale Behavioral Differences in Auschwitz-Birkenau with Simulation of Gendered Narratives},<br \/>\r\nauthor = {G\u00e1bor Mih\u00e1ly T\u00f3th and Tim Hempel and Krishna Somandepalli and Shrikanth Narayanan},<br \/>\r\nurl = {http:\/\/www.digitalhumanities.org\/dhq\/vol\/16\/3\/000622\/000622.html},<br \/>\r\nyear  = {2022},<br \/>\r\ndate = {2022-08-29},<br \/>\r\nurldate = {2022-08-29},<br \/>\r\njournal = {Digital Humanities Quarterly},<br \/>\r\nvolume = {16},<br \/>\r\nnumber = {3},<br \/>\r\nabstract = {In Auschwitz-Birkenau men and women were detained separately; anecdotal evidence suggests that they behaved differently. However, producing evidence based insights into victims' behavior is challenging. Perpetrators frequently destroyed camp documentations; victims' perspective remains dispersed in thousands of oral history interviews with survivors. Listening to, watching, or reading these thousands of interviews is not viable, and there is no established computational approach to gather systematic evidence from a large number of interviews. In this study, by applying methods and concepts of molecular physics, we developed a conceptual framework and computational approach to study thousands of human stories and we investigated 6628 interviews by survivors of the Auschwitz-Birkenau death camp. We applied the concept of state space and the Markov State Model to model the ensemble of 6628 testimonies. The Markov State Model along with the Transition Path Theory allowed us to compare the way women and men remember their time in the camp. We found that acts of solidarity and social bonds are the most important topics in their testimonies. However, we found that women are much more likely to address these topics. We provide systematic evidence that not only were women more likely to recall solidarity and social relations in their belated testimonies but they were also more likely to perform acts of solidarity and form social bonds in Auschwitz-Birkenau. Oral history interviews with Holocaust survivors constitute an important digital cultural heritage that documents one of the darkest moments in human history; generally, oral history collections are ubiquitous sources of modern history and significant assets of libraries and archives. We anticipate that our conceptual and computational framework will contribute not only to the understanding of gender behavior but also to the exploration of oral history as a cultural heritage, as well as to the computational study of narratives. This paper presents novel synergies between history, computer science, and physics, and it aims to stimulate further collaborations between these fields.},<br \/>\r\nkeywords = {Auschwitz, computational narrative modeling, survivor testimonies},<br \/>\r\npubstate = {published},<br \/>\r\ntppubtype = {article}<br \/>\r\n}<br \/>\r\n<\/pre><\/div><p class=\"tp_close_menu\"><a class=\"tp_close\" onclick=\"teachpress_pub_showhide('31','tp_bibtex')\">Close<\/a><\/p><\/div><div class=\"tp_abstract\" id=\"tp_abstract_31\" style=\"display:none;\"><div class=\"tp_abstract_entry\">In Auschwitz-Birkenau men and women were detained separately; anecdotal evidence suggests that they behaved differently. However, producing evidence based insights into victims&#039; behavior is challenging. Perpetrators frequently destroyed camp documentations; victims&#039; perspective remains dispersed in thousands of oral history interviews with survivors. Listening to, watching, or reading these thousands of interviews is not viable, and there is no established computational approach to gather systematic evidence from a large number of interviews. In this study, by applying methods and concepts of molecular physics, we developed a conceptual framework and computational approach to study thousands of human stories and we investigated 6628 interviews by survivors of the Auschwitz-Birkenau death camp. We applied the concept of state space and the Markov State Model to model the ensemble of 6628 testimonies. The Markov State Model along with the Transition Path Theory allowed us to compare the way women and men remember their time in the camp. We found that acts of solidarity and social bonds are the most important topics in their testimonies. However, we found that women are much more likely to address these topics. We provide systematic evidence that not only were women more likely to recall solidarity and social relations in their belated testimonies but they were also more likely to perform acts of solidarity and form social bonds in Auschwitz-Birkenau. Oral history interviews with Holocaust survivors constitute an important digital cultural heritage that documents one of the darkest moments in human history; generally, oral history collections are ubiquitous sources of modern history and significant assets of libraries and archives. We anticipate that our conceptual and computational framework will contribute not only to the understanding of gender behavior but also to the exploration of oral history as a cultural heritage, as well as to the computational study of narratives. This paper presents novel synergies between history, computer science, and physics, and it aims to stimulate further collaborations between these fields.<\/div><p class=\"tp_close_menu\"><a class=\"tp_close\" onclick=\"teachpress_pub_showhide('31','tp_abstract')\">Close<\/a><\/p><\/div><div class=\"tp_links\" id=\"tp_links_31\" style=\"display:none;\"><div class=\"tp_links_entry\"><ul class=\"tp_pub_list\"><li><i class=\"fas fa-globe\"><\/i><a class=\"tp_pub_list\" href=\"http:\/\/www.digitalhumanities.org\/dhq\/vol\/16\/3\/000622\/000622.html\" title=\"http:\/\/www.digitalhumanities.org\/dhq\/vol\/16\/3\/000622\/000622.html\" target=\"_blank\">http:\/\/www.digitalhumanities.org\/dhq\/vol\/16\/3\/000622\/000622.html<\/a><\/li><\/ul><\/div><p class=\"tp_close_menu\"><a class=\"tp_close\" onclick=\"teachpress_pub_showhide('31','tp_links')\">Close<\/a><\/p><\/div><\/div><\/div><div class=\"tp_publication tp_publication_article\"><div class=\"tp_pub_info\"><p class=\"tp_pub_author\"> Baruah, Sabyasachee;  Somandepalli, Krishna;  Narayanan, Shrikanth<\/p><p class=\"tp_pub_title\"><a class=\"tp_title_link\" onclick=\"teachpress_pub_showhide('22','tp_links')\" style=\"cursor:pointer;\">Representation of professions in entertainment media: Insights into frequency and sentiment trends through computational text analysis<\/a> <span class=\"tp_pub_type article\">Journal Article<\/span> <\/p><p class=\"tp_pub_additional\"><span class=\"tp_pub_additional_in\">In: <\/span><span class=\"tp_pub_additional_journal\">PLoS ONE, <\/span><span class=\"tp_pub_additional_year\">2022<\/span>.<\/p><p class=\"tp_pub_menu\"><span class=\"tp_abstract_link\"><a id=\"tp_abstract_sh_22\" class=\"tp_show\" onclick=\"teachpress_pub_showhide('22','tp_abstract')\" title=\"Show abstract\" style=\"cursor:pointer;\">Abstract<\/a><\/span> | <span class=\"tp_resource_link\"><a id=\"tp_links_sh_22\" class=\"tp_show\" onclick=\"teachpress_pub_showhide('22','tp_links')\" title=\"Show links and resources\" style=\"cursor:pointer;\">Links<\/a><\/span> | <span class=\"tp_bibtex_link\"><a id=\"tp_bibtex_sh_22\" class=\"tp_show\" onclick=\"teachpress_pub_showhide('22','tp_bibtex')\" title=\"Show BibTeX entry\" style=\"cursor:pointer;\">BibTeX<\/a><\/span> | <span class=\"tp_pub_tags_label\">Tags: <\/span><a rel=\"nofollow\" href=\"https:\/\/sail.usc.edu:\/ccmi\/publications\/?tgid=10#tppubs\" title=\"Show all publications which have a relationship to this tag\">Media<\/a>, <a rel=\"nofollow\" href=\"https:\/\/sail.usc.edu:\/ccmi\/publications\/?tgid=9#tppubs\" title=\"Show all publications which have a relationship to this tag\">Professions<\/a><\/p><div class=\"tp_bibtex\" id=\"tp_bibtex_22\" style=\"display:none;\"><div class=\"tp_bibtex_entry\"><pre>@article{baruah2021representation,<br \/>\r\ntitle = {Representation of professions in entertainment media: Insights into frequency and sentiment trends through computational text analysis},<br \/>\r\nauthor = {Sabyasachee Baruah and Krishna Somandepalli and Shrikanth Narayanan},<br \/>\r\nurl = {https:\/\/journals.plos.org\/plosone\/article?id=10.1371\/journal.pone.0267812},<br \/>\r\nyear  = {2022},<br \/>\r\ndate = {2022-05-18},<br \/>\r\njournal = {PLoS ONE},<br \/>\r\nabstract = {Societal ideas and trends dictate media narratives and cinematic depictions which in turn influence people\u2019s beliefs and perceptions of the real world. Media portrayal of individuals and social institutions related to culture, education, government, religion, and family affect their function and evolution over time as people perceive and incorporate the representations from portrayals into their everyday lives. It is important to study media depictions of social structures so that they do not propagate or reinforce negative stereotypes, or discriminate against a particular section of the society. In this work, we examine media representation of different professions and provide computational insights into their incidence, and sentiment expressed, in entertainment media content. We create a searchable taxonomy of professional groups, synsets, and titles to facilitate their retrieval from short-context speaker-agnostic text passages like movie and television (TV) show subtitles. We leverage this taxonomy and relevant natural language processing models to create a corpus of professional mentions in media content, spanning more than 136,000 IMDb titles over seven decades (1950-2017). We analyze the frequency and sentiment trends of different occupations, study the effect of media attributes such as genre, country of production, and title type on these trends, and investigate whether the incidence of professions in media subtitles correlate with their real-world employment statistics. We observe increased media mentions over time of STEM, arts, sports, and entertainment occupations in the analyzed subtitles, and a decreased frequency of manual labor jobs and military occupations. The sentiment expressed toward lawyers, police, and doctors showed increasing negative trends over time, whereas the mentions about astronauts, musicians, singers, and engineers appear more favorably. We found that genre is a good predictor of the type of professions mentioned in movies and TV shows. Professions that employ more people showed increased media frequency.},<br \/>\r\nkeywords = {Media, Professions},<br \/>\r\npubstate = {published},<br \/>\r\ntppubtype = {article}<br \/>\r\n}<br \/>\r\n<\/pre><\/div><p class=\"tp_close_menu\"><a class=\"tp_close\" onclick=\"teachpress_pub_showhide('22','tp_bibtex')\">Close<\/a><\/p><\/div><div class=\"tp_abstract\" id=\"tp_abstract_22\" style=\"display:none;\"><div class=\"tp_abstract_entry\">Societal ideas and trends dictate media narratives and cinematic depictions which in turn influence people\u2019s beliefs and perceptions of the real world. Media portrayal of individuals and social institutions related to culture, education, government, religion, and family affect their function and evolution over time as people perceive and incorporate the representations from portrayals into their everyday lives. It is important to study media depictions of social structures so that they do not propagate or reinforce negative stereotypes, or discriminate against a particular section of the society. In this work, we examine media representation of different professions and provide computational insights into their incidence, and sentiment expressed, in entertainment media content. We create a searchable taxonomy of professional groups, synsets, and titles to facilitate their retrieval from short-context speaker-agnostic text passages like movie and television (TV) show subtitles. We leverage this taxonomy and relevant natural language processing models to create a corpus of professional mentions in media content, spanning more than 136,000 IMDb titles over seven decades (1950-2017). We analyze the frequency and sentiment trends of different occupations, study the effect of media attributes such as genre, country of production, and title type on these trends, and investigate whether the incidence of professions in media subtitles correlate with their real-world employment statistics. We observe increased media mentions over time of STEM, arts, sports, and entertainment occupations in the analyzed subtitles, and a decreased frequency of manual labor jobs and military occupations. The sentiment expressed toward lawyers, police, and doctors showed increasing negative trends over time, whereas the mentions about astronauts, musicians, singers, and engineers appear more favorably. We found that genre is a good predictor of the type of professions mentioned in movies and TV shows. Professions that employ more people showed increased media frequency.<\/div><p class=\"tp_close_menu\"><a class=\"tp_close\" onclick=\"teachpress_pub_showhide('22','tp_abstract')\">Close<\/a><\/p><\/div><div class=\"tp_links\" id=\"tp_links_22\" style=\"display:none;\"><div class=\"tp_links_entry\"><ul class=\"tp_pub_list\"><li><i class=\"fas fa-globe\"><\/i><a class=\"tp_pub_list\" href=\"https:\/\/journals.plos.org\/plosone\/article?id=10.1371\/journal.pone.0267812\" title=\"https:\/\/journals.plos.org\/plosone\/article?id=10.1371\/journal.pone.0267812\" target=\"_blank\">https:\/\/journals.plos.org\/plosone\/article?id=10.1371\/journal.pone.0267812<\/a><\/li><\/ul><\/div><p class=\"tp_close_menu\"><a class=\"tp_close\" onclick=\"teachpress_pub_showhide('22','tp_links')\">Close<\/a><\/p><\/div><\/div><\/div><h3 class=\"tp_h3\" id=\"tp_h3_2021\">2021<\/h3><div class=\"tp_publication tp_publication_inproceedings\"><div class=\"tp_pub_info\"><p class=\"tp_pub_author\"> Baruah, Sabyasachee;  Chakravarthula, Sandeep Nallan;  Narayanan, Shrikanth<\/p><p class=\"tp_pub_title\"><a class=\"tp_title_link\" onclick=\"teachpress_pub_showhide('20','tp_links')\" style=\"cursor:pointer;\">Annotation and Evaluation of Coreference Resolution in Screenplays<\/a> <span class=\"tp_pub_type inproceedings\">Inproceedings<\/span> <\/p><p class=\"tp_pub_additional\"><span class=\"tp_pub_additional_in\">In: <\/span><span class=\"tp_pub_additional_pages\">pp. 2004\u20132010, <\/span><span class=\"tp_pub_additional_publisher\">Association for Computational Linguistics, <\/span><span class=\"tp_pub_additional_year\">2021<\/span>.<\/p><p class=\"tp_pub_menu\"><span class=\"tp_abstract_link\"><a id=\"tp_abstract_sh_20\" class=\"tp_show\" onclick=\"teachpress_pub_showhide('20','tp_abstract')\" title=\"Show abstract\" style=\"cursor:pointer;\">Abstract<\/a><\/span> | <span class=\"tp_resource_link\"><a id=\"tp_links_sh_20\" class=\"tp_show\" onclick=\"teachpress_pub_showhide('20','tp_links')\" title=\"Show links and resources\" style=\"cursor:pointer;\">Links<\/a><\/span> | <span class=\"tp_bibtex_link\"><a id=\"tp_bibtex_sh_20\" class=\"tp_show\" onclick=\"teachpress_pub_showhide('20','tp_bibtex')\" title=\"Show BibTeX entry\" style=\"cursor:pointer;\">BibTeX<\/a><\/span> | <span class=\"tp_pub_tags_label\">Tags: <\/span><a rel=\"nofollow\" href=\"https:\/\/sail.usc.edu:\/ccmi\/publications\/?tgid=4#tppubs\" title=\"Show all publications which have a relationship to this tag\">coreference resolution<\/a><\/p><div class=\"tp_bibtex\" id=\"tp_bibtex_20\" style=\"display:none;\"><div class=\"tp_bibtex_entry\"><pre>@inproceedings{baruah-etal-2021-annotation,<br \/>\r\ntitle = {Annotation and Evaluation of Coreference Resolution in Screenplays},<br \/>\r\nauthor = {Sabyasachee Baruah and Sandeep Nallan Chakravarthula and Shrikanth Narayanan},<br \/>\r\ndoi = {10.18653\/v1\/2021.findings-acl.176},<br \/>\r\nyear  = {2021},<br \/>\r\ndate = {2021-08-02},<br \/>\r\npages = {2004\u20132010},<br \/>\r\npublisher = {Association for Computational Linguistics},<br \/>\r\nabstract = {Screenplays refer to characters using different names, pronouns, and nominal expressions. We need to resolve these mentions to<br \/>\r\nthe correct referent character for better story understanding and holistic research in computational narratology. Coreference resolution of character mentions in screenplays becomes<br \/>\r\nchallenging because of the large document lengths, unique structural features like scene headers, interleaving of action and speech<br \/>\r\npassages, and reliance on the accompanying video. In this work, we first adapt widely used annotation guidelines to address domain-specific issues in screenplays. We develop an automatic screenplay parser to extract the<br \/>\r\nstructural information and design coreference rules based upon the structure. Our model exploits these structural features and outperforms a benchmark coreference model on the<br \/>\r\nscreenplay coreference resolution task.},<br \/>\r\nkeywords = {coreference resolution},<br \/>\r\npubstate = {published},<br \/>\r\ntppubtype = {inproceedings}<br \/>\r\n}<br \/>\r\n<\/pre><\/div><p class=\"tp_close_menu\"><a class=\"tp_close\" onclick=\"teachpress_pub_showhide('20','tp_bibtex')\">Close<\/a><\/p><\/div><div class=\"tp_abstract\" id=\"tp_abstract_20\" style=\"display:none;\"><div class=\"tp_abstract_entry\">Screenplays refer to characters using different names, pronouns, and nominal expressions. We need to resolve these mentions to<br \/>\r\nthe correct referent character for better story understanding and holistic research in computational narratology. Coreference resolution of character mentions in screenplays becomes<br \/>\r\nchallenging because of the large document lengths, unique structural features like scene headers, interleaving of action and speech<br \/>\r\npassages, and reliance on the accompanying video. In this work, we first adapt widely used annotation guidelines to address domain-specific issues in screenplays. We develop an automatic screenplay parser to extract the<br \/>\r\nstructural information and design coreference rules based upon the structure. Our model exploits these structural features and outperforms a benchmark coreference model on the<br \/>\r\nscreenplay coreference resolution task.<\/div><p class=\"tp_close_menu\"><a class=\"tp_close\" onclick=\"teachpress_pub_showhide('20','tp_abstract')\">Close<\/a><\/p><\/div><div class=\"tp_links\" id=\"tp_links_20\" style=\"display:none;\"><div class=\"tp_links_entry\"><ul class=\"tp_pub_list\"><li><i class=\"ai ai-doi\"><\/i><a class=\"tp_pub_list\" href=\"https:\/\/dx.doi.org\/10.18653\/v1\/2021.findings-acl.176\" title=\"Follow DOI:10.18653\/v1\/2021.findings-acl.176\" target=\"_blank\">doi:10.18653\/v1\/2021.findings-acl.176<\/a><\/li><\/ul><\/div><p class=\"tp_close_menu\"><a class=\"tp_close\" onclick=\"teachpress_pub_showhide('20','tp_links')\">Close<\/a><\/p><\/div><\/div><\/div><div class=\"tp_publication tp_publication_article\"><div class=\"tp_pub_info\"><p class=\"tp_pub_author\"> Somandepalli, Krishna;  Hebbar, Rajat;  Narayanan, Shrikanth S<\/p><p class=\"tp_pub_title\"><a class=\"tp_title_link\" onclick=\"teachpress_pub_showhide('1','tp_links')\" style=\"cursor:pointer;\">Multi-Face: Self-supervised Multiview Adaptation for Robust Face Clustering in Videos<\/a> <span class=\"tp_pub_type article\">Journal Article<\/span> <\/p><p class=\"tp_pub_additional\"><span class=\"tp_pub_additional_in\">In: <\/span><span class=\"tp_pub_additional_journal\">IEEE Transactions on Multimedia, <\/span><span class=\"tp_pub_additional_year\">2021<\/span>, <span class=\"tp_pub_additional_issn\">ISSN: 1520-9210<\/span>.<\/p><p class=\"tp_pub_menu\"><span class=\"tp_abstract_link\"><a id=\"tp_abstract_sh_1\" class=\"tp_show\" onclick=\"teachpress_pub_showhide('1','tp_abstract')\" title=\"Show abstract\" style=\"cursor:pointer;\">Abstract<\/a><\/span> | <span class=\"tp_resource_link\"><a id=\"tp_links_sh_1\" class=\"tp_show\" onclick=\"teachpress_pub_showhide('1','tp_links')\" title=\"Show links and resources\" style=\"cursor:pointer;\">Links<\/a><\/span> | <span class=\"tp_bibtex_link\"><a id=\"tp_bibtex_sh_1\" class=\"tp_show\" onclick=\"teachpress_pub_showhide('1','tp_bibtex')\" title=\"Show BibTeX entry\" style=\"cursor:pointer;\">BibTeX<\/a><\/span> | <span class=\"tp_pub_tags_label\">Tags: <\/span><\/p><div class=\"tp_bibtex\" id=\"tp_bibtex_1\" style=\"display:none;\"><div class=\"tp_bibtex_entry\"><pre>@article{Somandepalli2020MultiFaceSM,<br \/>\r\ntitle = {Multi-Face: Self-supervised Multiview Adaptation for Robust Face Clustering in Videos},<br \/>\r\nauthor = {Krishna Somandepalli and Rajat Hebbar and Shrikanth S Narayanan},<br \/>\r\ndoi = {10.1109\/TMM.2021.3096155},<br \/>\r\nissn = {1520-9210},<br \/>\r\nyear  = {2021},<br \/>\r\ndate = {2021-07-09},<br \/>\r\njournal = {IEEE Transactions on Multimedia},<br \/>\r\nabstract = {Robust face clustering is a vital step in enabling a computational understanding of visual character portrayal in media. Face clustering for long-form content such as movies is challenging because of variations in appearance and lack of large-scale labeled data resources. Our work focuses on two key aspects of this problem: the lack of domain-specific training or benchmark datasets and adapting face embeddings learned on web images to the domain of movie videos. First, we curated over 169,000 face tracks from 240 Hollywood movies with weak labels on whether a pair of face tracks belong to the same or different characters. We proposed an offline nearest-neighbor search in the embedding space to mine hard-examples from these tracks. We then explored triplet-loss and multiview correlation-based methods for adapting face image embeddings to hard-examples from movie videos. We also developed SAIL-Movie Character Benchmark corpus to augment existing benchmarks with more racially diverse characters and provided face-quality labels for subsequent error analysis. Our experimental results highlight the use of weakly labeled data for domain-specific feature adaptation. Overall, we found that multiview correlation-based adaptation yielded robust and more discriminative face embeddings. Its performance on downstream face verification and clustering tasks was comparable to that of the state-of-the-art results in this domain. We hope that the large-scale datasets developed in this work can further advance automatic character labeling in videos. All resources are available at https:\/\/sail.usc.edu\/~ccmi\/multiface.},<br \/>\r\nkeywords = {},<br \/>\r\npubstate = {published},<br \/>\r\ntppubtype = {article}<br \/>\r\n}<br \/>\r\n<\/pre><\/div><p class=\"tp_close_menu\"><a class=\"tp_close\" onclick=\"teachpress_pub_showhide('1','tp_bibtex')\">Close<\/a><\/p><\/div><div class=\"tp_abstract\" id=\"tp_abstract_1\" style=\"display:none;\"><div class=\"tp_abstract_entry\">Robust face clustering is a vital step in enabling a computational understanding of visual character portrayal in media. Face clustering for long-form content such as movies is challenging because of variations in appearance and lack of large-scale labeled data resources. Our work focuses on two key aspects of this problem: the lack of domain-specific training or benchmark datasets and adapting face embeddings learned on web images to the domain of movie videos. First, we curated over 169,000 face tracks from 240 Hollywood movies with weak labels on whether a pair of face tracks belong to the same or different characters. We proposed an offline nearest-neighbor search in the embedding space to mine hard-examples from these tracks. We then explored triplet-loss and multiview correlation-based methods for adapting face image embeddings to hard-examples from movie videos. We also developed SAIL-Movie Character Benchmark corpus to augment existing benchmarks with more racially diverse characters and provided face-quality labels for subsequent error analysis. Our experimental results highlight the use of weakly labeled data for domain-specific feature adaptation. Overall, we found that multiview correlation-based adaptation yielded robust and more discriminative face embeddings. Its performance on downstream face verification and clustering tasks was comparable to that of the state-of-the-art results in this domain. We hope that the large-scale datasets developed in this work can further advance automatic character labeling in videos. All resources are available at https:\/\/sail.usc.edu\/~ccmi\/multiface.<\/div><p class=\"tp_close_menu\"><a class=\"tp_close\" onclick=\"teachpress_pub_showhide('1','tp_abstract')\">Close<\/a><\/p><\/div><div class=\"tp_links\" id=\"tp_links_1\" style=\"display:none;\"><div class=\"tp_links_entry\"><ul class=\"tp_pub_list\"><li><i class=\"ai ai-doi\"><\/i><a class=\"tp_pub_list\" href=\"https:\/\/dx.doi.org\/10.1109\/TMM.2021.3096155\" title=\"Follow DOI:10.1109\/TMM.2021.3096155\" target=\"_blank\">doi:10.1109\/TMM.2021.3096155<\/a><\/li><\/ul><\/div><p class=\"tp_close_menu\"><a class=\"tp_close\" onclick=\"teachpress_pub_showhide('1','tp_links')\">Close<\/a><\/p><\/div><\/div><\/div><div class=\"tp_publication tp_publication_article\"><div class=\"tp_pub_info\"><p class=\"tp_pub_author\"> Somandepalli, Krishna;  Hebbar, Rajat;  Narayanan, Shrikanth<\/p><p class=\"tp_pub_title\"><a class=\"tp_title_link\" onclick=\"teachpress_pub_showhide('32','tp_links')\" style=\"cursor:pointer;\">Robust Character Labeling in Movie Videos: Data Resources and Self-supervised Feature Adaptation.<\/a> <span class=\"tp_pub_type article\">Journal Article<\/span> <\/p><p class=\"tp_pub_additional\"><span class=\"tp_pub_additional_in\">In: <\/span><span class=\"tp_pub_additional_journal\">IEEE Transactions on Multimedia, <\/span><span class=\"tp_pub_additional_volume\">24 <\/span>, <span class=\"tp_pub_additional_pages\">pp. 3355 - 3368, <\/span><span class=\"tp_pub_additional_year\">2021<\/span>.<\/p><p class=\"tp_pub_menu\"><span class=\"tp_abstract_link\"><a id=\"tp_abstract_sh_32\" class=\"tp_show\" onclick=\"teachpress_pub_showhide('32','tp_abstract')\" title=\"Show abstract\" style=\"cursor:pointer;\">Abstract<\/a><\/span> | <span class=\"tp_resource_link\"><a id=\"tp_links_sh_32\" class=\"tp_show\" onclick=\"teachpress_pub_showhide('32','tp_links')\" title=\"Show links and resources\" style=\"cursor:pointer;\">Links<\/a><\/span> | <span class=\"tp_bibtex_link\"><a id=\"tp_bibtex_sh_32\" class=\"tp_show\" onclick=\"teachpress_pub_showhide('32','tp_bibtex')\" title=\"Show BibTeX entry\" style=\"cursor:pointer;\">BibTeX<\/a><\/span> | <span class=\"tp_pub_tags_label\">Tags: <\/span><a rel=\"nofollow\" href=\"https:\/\/sail.usc.edu:\/ccmi\/publications\/?tgid=22#tppubs\" title=\"Show all publications which have a relationship to this tag\">computational media understanding<\/a>, <a rel=\"nofollow\" href=\"https:\/\/sail.usc.edu:\/ccmi\/publications\/?tgid=21#tppubs\" title=\"Show all publications which have a relationship to this tag\">face clustering<\/a>, <a rel=\"nofollow\" href=\"https:\/\/sail.usc.edu:\/ccmi\/publications\/?tgid=20#tppubs\" title=\"Show all publications which have a relationship to this tag\">face diarization<\/a>, <a rel=\"nofollow\" href=\"https:\/\/sail.usc.edu:\/ccmi\/publications\/?tgid=18#tppubs\" title=\"Show all publications which have a relationship to this tag\">multiview correlation<\/a>, <a rel=\"nofollow\" href=\"https:\/\/sail.usc.edu:\/ccmi\/publications\/?tgid=17#tppubs\" title=\"Show all publications which have a relationship to this tag\">self-supervision<\/a>, <a rel=\"nofollow\" href=\"https:\/\/sail.usc.edu:\/ccmi\/publications\/?tgid=19#tppubs\" title=\"Show all publications which have a relationship to this tag\">triplet loss<\/a>, <a rel=\"nofollow\" href=\"https:\/\/sail.usc.edu:\/ccmi\/publications\/?tgid=16#tppubs\" title=\"Show all publications which have a relationship to this tag\">video character labeling<\/a><\/p><div class=\"tp_bibtex\" id=\"tp_bibtex_32\" style=\"display:none;\"><div class=\"tp_bibtex_entry\"><pre>@article{Somandepalli2021b,<br \/>\r\ntitle = {Robust Character Labeling in Movie Videos: Data Resources and Self-supervised Feature Adaptation.},<br \/>\r\nauthor = {Krishna Somandepalli and Rajat Hebbar and Shrikanth Narayanan},<br \/>\r\nurl = {https:\/\/sail.usc.edu\/publications\/files\/Somandepalli-TMM2021.pdf},<br \/>\r\ndoi = {10.1109\/TMM.2021.3096155},<br \/>\r\nyear  = {2021},<br \/>\r\ndate = {2021-07-09},<br \/>\r\nurldate = {2021-07-09},<br \/>\r\njournal = {IEEE Transactions on Multimedia},<br \/>\r\nvolume = {24},<br \/>\r\npages = {3355 - 3368},<br \/>\r\nabstract = {Robust face clustering is a vital step in enabling computational understanding of visual character portrayal in media. Face clustering for long-form content is challenging because of variations in appearance and lack of supporting large-scale labeled data. Our work in this paper focuses on two key aspects of this problem: the lack of domain-specific training or benchmark datasets, and adapting face embeddings learned on web images to long-form content, specifically movies. First, we present a dataset of over 169000 face tracks curated from 240 Hollywood movies with weak labels on whether a pair of face tracks belong to the same or a different character. We propose an offline algorithm based on nearest-neighbor search in the embedding space to mine hard-examples from these tracks. We then investigate triplet-loss and multiview correlation-based methods for adapting face embeddings to hard-examples. Our experimental results highlight the usefulness of weakly labeled data for domain-specific feature adaptation. Overall, we find that multiview correlation-based adaptation yields more discriminative and robust face embeddings. Its performance on downstream face verification and clustering tasks is comparable to that of the state-of-the-art results in this domain. We also present the SAIL-Movie Character Benchmark corpus developed to augment existing benchmarks. It consists of racially diverse actors and provides face-quality labels for subsequent error analysis. We hope that the large-scale datasets developed in this work can further advance automatic character labeling in videos. All resources are available freely at https:\/\/sail.usc.edu\/~ccmi\/multiface .<br \/>\r\n},<br \/>\r\nkeywords = {computational media understanding, face clustering, face diarization, multiview correlation, self-supervision, triplet loss, video character labeling},<br \/>\r\npubstate = {published},<br \/>\r\ntppubtype = {article}<br \/>\r\n}<br \/>\r\n<\/pre><\/div><p class=\"tp_close_menu\"><a class=\"tp_close\" onclick=\"teachpress_pub_showhide('32','tp_bibtex')\">Close<\/a><\/p><\/div><div class=\"tp_abstract\" id=\"tp_abstract_32\" style=\"display:none;\"><div class=\"tp_abstract_entry\">Robust face clustering is a vital step in enabling computational understanding of visual character portrayal in media. Face clustering for long-form content is challenging because of variations in appearance and lack of supporting large-scale labeled data. Our work in this paper focuses on two key aspects of this problem: the lack of domain-specific training or benchmark datasets, and adapting face embeddings learned on web images to long-form content, specifically movies. First, we present a dataset of over 169000 face tracks curated from 240 Hollywood movies with weak labels on whether a pair of face tracks belong to the same or a different character. We propose an offline algorithm based on nearest-neighbor search in the embedding space to mine hard-examples from these tracks. We then investigate triplet-loss and multiview correlation-based methods for adapting face embeddings to hard-examples. Our experimental results highlight the usefulness of weakly labeled data for domain-specific feature adaptation. Overall, we find that multiview correlation-based adaptation yields more discriminative and robust face embeddings. Its performance on downstream face verification and clustering tasks is comparable to that of the state-of-the-art results in this domain. We also present the SAIL-Movie Character Benchmark corpus developed to augment existing benchmarks. It consists of racially diverse actors and provides face-quality labels for subsequent error analysis. We hope that the large-scale datasets developed in this work can further advance automatic character labeling in videos. All resources are available freely at https:\/\/sail.usc.edu\/~ccmi\/multiface .<br \/>\r\n<\/div><p class=\"tp_close_menu\"><a class=\"tp_close\" onclick=\"teachpress_pub_showhide('32','tp_abstract')\">Close<\/a><\/p><\/div><div class=\"tp_links\" id=\"tp_links_32\" style=\"display:none;\"><div class=\"tp_links_entry\"><ul class=\"tp_pub_list\"><li><i class=\"fas fa-file-pdf\"><\/i><a class=\"tp_pub_list\" href=\"https:\/\/sail.usc.edu\/publications\/files\/Somandepalli-TMM2021.pdf\" title=\"https:\/\/sail.usc.edu\/publications\/files\/Somandepalli-TMM2021.pdf\" target=\"_blank\">https:\/\/sail.usc.edu\/publications\/files\/Somandepalli-TMM2021.pdf<\/a><\/li><li><i class=\"ai ai-doi\"><\/i><a class=\"tp_pub_list\" href=\"https:\/\/dx.doi.org\/10.1109\/TMM.2021.3096155\" title=\"Follow DOI:10.1109\/TMM.2021.3096155\" target=\"_blank\">doi:10.1109\/TMM.2021.3096155<\/a><\/li><\/ul><\/div><p class=\"tp_close_menu\"><a class=\"tp_close\" onclick=\"teachpress_pub_showhide('32','tp_links')\">Close<\/a><\/p><\/div><\/div><\/div><div class=\"tp_publication tp_publication_inproceedings\"><div class=\"tp_pub_info\"><p class=\"tp_pub_author\"> Hebbar, Rajat;  Somandepalli, Krishna;  Peri, Raghuveer;  Travadi, Ruchir;  Tuplin, Tracy;  Rivera, Fernando;  Narayanan, Shrikanth<\/p><p class=\"tp_pub_title\">A Computational Tool to Study Vocal Participation of Women in UN-ITU Meetings <span class=\"tp_pub_type inproceedings\">Inproceedings<\/span> <\/p><p class=\"tp_pub_additional\"><span class=\"tp_pub_additional_in\">In: <\/span><span class=\"tp_pub_additional_booktitle\">2021 International Conference on Content-Based Multimedia Indexing (CBMI), <\/span><span class=\"tp_pub_additional_pages\">pp. 1\u20134, <\/span><span class=\"tp_pub_additional_organization\">IEEE <\/span><span class=\"tp_pub_additional_year\">2021<\/span>.<\/p><p class=\"tp_pub_menu\"><span class=\"tp_bibtex_link\"><a id=\"tp_bibtex_sh_2\" class=\"tp_show\" onclick=\"teachpress_pub_showhide('2','tp_bibtex')\" title=\"Show BibTeX entry\" style=\"cursor:pointer;\">BibTeX<\/a><\/span> | <span class=\"tp_pub_tags_label\">Tags: <\/span><\/p><div class=\"tp_bibtex\" id=\"tp_bibtex_2\" style=\"display:none;\"><div class=\"tp_bibtex_entry\"><pre>@inproceedings{hebbar2021computational,<br \/>\r\ntitle = {A Computational Tool to Study Vocal Participation of Women in UN-ITU Meetings},<br \/>\r\nauthor = {Rajat Hebbar and Krishna Somandepalli and Raghuveer Peri and Ruchir Travadi and Tracy Tuplin and Fernando Rivera and Shrikanth Narayanan},<br \/>\r\nyear  = {2021},<br \/>\r\ndate = {2021-01-01},<br \/>\r\nbooktitle = {2021 International Conference on Content-Based Multimedia Indexing (CBMI)},<br \/>\r\npages = {1--4},<br \/>\r\norganization = {IEEE},<br \/>\r\nkeywords = {},<br \/>\r\npubstate = {published},<br \/>\r\ntppubtype = {inproceedings}<br \/>\r\n}<br \/>\r\n<\/pre><\/div><p class=\"tp_close_menu\"><a class=\"tp_close\" onclick=\"teachpress_pub_showhide('2','tp_bibtex')\">Close<\/a><\/p><\/div><\/div><\/div><div class=\"tp_publication tp_publication_inproceedings\"><div class=\"tp_pub_info\"><p class=\"tp_pub_author\"> Knox, Dillon;  Greer, Timothy;  Ma, Benjamin;  Kuo, Emily;  Somandepalli, Krishna;  Narayanan, Shrikanth<\/p><p class=\"tp_pub_title\">Loss Function Approaches for Multi-label Music Tagging <span class=\"tp_pub_type inproceedings\">Inproceedings<\/span> <\/p><p class=\"tp_pub_additional\"><span class=\"tp_pub_additional_in\">In: <\/span><span class=\"tp_pub_additional_booktitle\">2021 International Conference on Content-Based Multimedia Indexing (CBMI), <\/span><span class=\"tp_pub_additional_pages\">pp. 1\u20134, <\/span><span class=\"tp_pub_additional_organization\">IEEE <\/span><span class=\"tp_pub_additional_year\">2021<\/span>.<\/p><p class=\"tp_pub_menu\"><span class=\"tp_bibtex_link\"><a id=\"tp_bibtex_sh_3\" class=\"tp_show\" onclick=\"teachpress_pub_showhide('3','tp_bibtex')\" title=\"Show BibTeX entry\" style=\"cursor:pointer;\">BibTeX<\/a><\/span> | <span class=\"tp_pub_tags_label\">Tags: <\/span><\/p><div class=\"tp_bibtex\" id=\"tp_bibtex_3\" style=\"display:none;\"><div class=\"tp_bibtex_entry\"><pre>@inproceedings{knox2021loss,<br \/>\r\ntitle = {Loss Function Approaches for Multi-label Music Tagging},<br \/>\r\nauthor = {Dillon Knox and Timothy Greer and Benjamin Ma and Emily Kuo and Krishna Somandepalli and Shrikanth Narayanan},<br \/>\r\nyear  = {2021},<br \/>\r\ndate = {2021-01-01},<br \/>\r\nbooktitle = {2021 International Conference on Content-Based Multimedia Indexing (CBMI)},<br \/>\r\npages = {1--4},<br \/>\r\norganization = {IEEE},<br \/>\r\nkeywords = {},<br \/>\r\npubstate = {published},<br \/>\r\ntppubtype = {inproceedings}<br \/>\r\n}<br \/>\r\n<\/pre><\/div><p class=\"tp_close_menu\"><a class=\"tp_close\" onclick=\"teachpress_pub_showhide('3','tp_bibtex')\">Close<\/a><\/p><\/div><\/div><\/div><div class=\"tp_publication tp_publication_article\"><div class=\"tp_pub_info\"><p class=\"tp_pub_author\"> Ma, Benjamin;  Greer, Timothy;  Knox, Dillon;  Narayanan, Shrikanth<\/p><p class=\"tp_pub_title\">A computational lens into how music characterizes genre in film <span class=\"tp_pub_type article\">Journal Article<\/span> <\/p><p class=\"tp_pub_additional\"><span class=\"tp_pub_additional_in\">In: <\/span><span class=\"tp_pub_additional_journal\">PloS one, <\/span><span class=\"tp_pub_additional_volume\">16 <\/span><span class=\"tp_pub_additional_number\">(4), <\/span><span class=\"tp_pub_additional_pages\">pp. e0249957, <\/span><span class=\"tp_pub_additional_year\">2021<\/span>.<\/p><p class=\"tp_pub_menu\"><span class=\"tp_bibtex_link\"><a id=\"tp_bibtex_sh_4\" class=\"tp_show\" onclick=\"teachpress_pub_showhide('4','tp_bibtex')\" title=\"Show BibTeX entry\" style=\"cursor:pointer;\">BibTeX<\/a><\/span> | <span class=\"tp_pub_tags_label\">Tags: <\/span><\/p><div class=\"tp_bibtex\" id=\"tp_bibtex_4\" style=\"display:none;\"><div class=\"tp_bibtex_entry\"><pre>@article{ma2021computational,<br \/>\r\ntitle = {A computational lens into how music characterizes genre in film},<br \/>\r\nauthor = {Benjamin Ma and Timothy Greer and Dillon Knox and Shrikanth Narayanan},<br \/>\r\nyear  = {2021},<br \/>\r\ndate = {2021-01-01},<br \/>\r\njournal = {PloS one},<br \/>\r\nvolume = {16},<br \/>\r\nnumber = {4},<br \/>\r\npages = {e0249957},<br \/>\r\npublisher = {Public Library of Science San Francisco, CA USA},<br \/>\r\nkeywords = {},<br \/>\r\npubstate = {published},<br \/>\r\ntppubtype = {article}<br \/>\r\n}<br \/>\r\n<\/pre><\/div><p class=\"tp_close_menu\"><a class=\"tp_close\" onclick=\"teachpress_pub_showhide('4','tp_bibtex')\">Close<\/a><\/p><\/div><\/div><\/div><div class=\"tp_publication tp_publication_article\"><div class=\"tp_pub_info\"><p class=\"tp_pub_author\"> Somandepalli, Krishna;  Guha, Tanaya;  Martinez, Victor R;  Kumar, Naveen;  Adam, Hartwig;  Narayanan, Shrikanth<\/p><p class=\"tp_pub_title\">Computational media intelligence: human-centered machine analysis of media <span class=\"tp_pub_type article\">Journal Article<\/span> <\/p><p class=\"tp_pub_additional\"><span class=\"tp_pub_additional_in\">In: <\/span><span class=\"tp_pub_additional_journal\">Proceedings of the IEEE, <\/span><span class=\"tp_pub_additional_year\">2021<\/span>.<\/p><p class=\"tp_pub_menu\"><span class=\"tp_bibtex_link\"><a id=\"tp_bibtex_sh_5\" class=\"tp_show\" onclick=\"teachpress_pub_showhide('5','tp_bibtex')\" title=\"Show BibTeX entry\" style=\"cursor:pointer;\">BibTeX<\/a><\/span> | <span class=\"tp_pub_tags_label\">Tags: <\/span><\/p><div class=\"tp_bibtex\" id=\"tp_bibtex_5\" style=\"display:none;\"><div class=\"tp_bibtex_entry\"><pre>@article{somandepalli2021computational,<br \/>\r\ntitle = {Computational media intelligence: human-centered machine analysis of media},<br \/>\r\nauthor = {Krishna Somandepalli and Tanaya Guha and Victor R Martinez and Naveen Kumar and Hartwig Adam and Shrikanth Narayanan},<br \/>\r\nyear  = {2021},<br \/>\r\ndate = {2021-01-01},<br \/>\r\njournal = {Proceedings of the IEEE},<br \/>\r\npublisher = {IEEE},<br \/>\r\nkeywords = {},<br \/>\r\npubstate = {published},<br \/>\r\ntppubtype = {article}<br \/>\r\n}<br \/>\r\n<\/pre><\/div><p class=\"tp_close_menu\"><a class=\"tp_close\" onclick=\"teachpress_pub_showhide('5','tp_bibtex')\">Close<\/a><\/p><\/div><\/div><\/div><h3 class=\"tp_h3\" id=\"tp_h3_2020\">2020<\/h3><div class=\"tp_publication tp_publication_inproceedings\"><div class=\"tp_pub_info\"><p class=\"tp_pub_author\"> Martinez, Victor;  Somandepalli, Krishna;  Tehranian-Uhls, Yalda;  Narayanan, Shrikanth<\/p><p class=\"tp_pub_title\">Joint Estimation and Analysis of Risk Behavior Ratings in Movie Scripts <span class=\"tp_pub_type inproceedings\">Inproceedings<\/span> <\/p><p class=\"tp_pub_additional\"><span class=\"tp_pub_additional_in\">In: <\/span><span class=\"tp_pub_additional_booktitle\">Proceedings of the 2020 Conference on Empirical Methods in Natural Language Processing (EMNLP), <\/span><span class=\"tp_pub_additional_pages\">pp. 4780\u20134790, <\/span><span class=\"tp_pub_additional_year\">2020<\/span>.<\/p><p class=\"tp_pub_menu\"><span class=\"tp_bibtex_link\"><a id=\"tp_bibtex_sh_13\" class=\"tp_show\" onclick=\"teachpress_pub_showhide('13','tp_bibtex')\" title=\"Show BibTeX entry\" style=\"cursor:pointer;\">BibTeX<\/a><\/span> | <span class=\"tp_pub_tags_label\">Tags: <\/span><\/p><div class=\"tp_bibtex\" id=\"tp_bibtex_13\" style=\"display:none;\"><div class=\"tp_bibtex_entry\"><pre>@inproceedings{martinez2020joint,<br \/>\r\ntitle = {Joint Estimation and Analysis of Risk Behavior Ratings in Movie Scripts},<br \/>\r\nauthor = {Victor Martinez and Krishna Somandepalli and Yalda Tehranian-Uhls and Shrikanth Narayanan},<br \/>\r\nyear  = {2020},<br \/>\r\ndate = {2020-01-01},<br \/>\r\nbooktitle = {Proceedings of the 2020 Conference on Empirical Methods in Natural Language Processing (EMNLP)},<br \/>\r\npages = {4780--4790},<br \/>\r\nkeywords = {},<br \/>\r\npubstate = {published},<br \/>\r\ntppubtype = {inproceedings}<br \/>\r\n}<br \/>\r\n<\/pre><\/div><p class=\"tp_close_menu\"><a class=\"tp_close\" onclick=\"teachpress_pub_showhide('13','tp_bibtex')\">Close<\/a><\/p><\/div><\/div><\/div><div class=\"tp_publication tp_publication_article\"><div class=\"tp_pub_info\"><p class=\"tp_pub_author\"> Ramakrishna, Anil Kumar;  Gupta, Rahul;  Narayanan, Shrikanth<\/p><p class=\"tp_pub_title\">Joint Multi-Dimensional Model for Global and Time-Series Annotations <span class=\"tp_pub_type article\">Journal Article<\/span> <\/p><p class=\"tp_pub_additional\"><span class=\"tp_pub_additional_in\">In: <\/span><span class=\"tp_pub_additional_journal\">IEEE Transactions on Affective Computing, <\/span><span class=\"tp_pub_additional_year\">2020<\/span>.<\/p><p class=\"tp_pub_menu\"><span class=\"tp_bibtex_link\"><a id=\"tp_bibtex_sh_7\" class=\"tp_show\" onclick=\"teachpress_pub_showhide('7','tp_bibtex')\" title=\"Show BibTeX entry\" style=\"cursor:pointer;\">BibTeX<\/a><\/span> | <span class=\"tp_pub_tags_label\">Tags: <\/span><\/p><div class=\"tp_bibtex\" id=\"tp_bibtex_7\" style=\"display:none;\"><div class=\"tp_bibtex_entry\"><pre>@article{ramakrishna2020joint,<br \/>\r\ntitle = {Joint Multi-Dimensional Model for Global and Time-Series Annotations},<br \/>\r\nauthor = {Anil Kumar Ramakrishna and Rahul Gupta and Shrikanth Narayanan},<br \/>\r\nyear  = {2020},<br \/>\r\ndate = {2020-01-01},<br \/>\r\njournal = {IEEE Transactions on Affective Computing},<br \/>\r\npublisher = {IEEE},<br \/>\r\nkeywords = {},<br \/>\r\npubstate = {published},<br \/>\r\ntppubtype = {article}<br \/>\r\n}<br \/>\r\n<\/pre><\/div><p class=\"tp_close_menu\"><a class=\"tp_close\" onclick=\"teachpress_pub_showhide('7','tp_bibtex')\">Close<\/a><\/p><\/div><\/div><\/div><div class=\"tp_publication tp_publication_article\"><div class=\"tp_pub_info\"><p class=\"tp_pub_author\"> Narayanan, Shrikanth S;  Madni, Asad M<\/p><p class=\"tp_pub_title\">Inclusive Human centered Machine Intelligence <span class=\"tp_pub_type article\">Journal Article<\/span> <\/p><p class=\"tp_pub_additional\"><span class=\"tp_pub_additional_in\">In: <\/span><span class=\"tp_pub_additional_journal\">The Bridge, <\/span><span class=\"tp_pub_additional_volume\">50 <\/span>, <span class=\"tp_pub_additional_pages\">pp. 113-116, <\/span><span class=\"tp_pub_additional_year\">2020<\/span>.<\/p><p class=\"tp_pub_menu\"><span class=\"tp_bibtex_link\"><a id=\"tp_bibtex_sh_21\" class=\"tp_show\" onclick=\"teachpress_pub_showhide('21','tp_bibtex')\" title=\"Show BibTeX entry\" style=\"cursor:pointer;\">BibTeX<\/a><\/span> | <span class=\"tp_pub_tags_label\">Tags: <\/span><\/p><div class=\"tp_bibtex\" id=\"tp_bibtex_21\" style=\"display:none;\"><div class=\"tp_bibtex_entry\"><pre>@article{NarayananMadni-Bridge2020,<br \/>\r\ntitle = {Inclusive Human centered Machine Intelligence},<br \/>\r\nauthor = {Shrikanth S Narayanan and Asad M Madni},<br \/>\r\nyear  = {2020},<br \/>\r\ndate = {2020-01-01},<br \/>\r\njournal = {The Bridge},<br \/>\r\nvolume = {50},<br \/>\r\npages = {113-116},<br \/>\r\npublisher = {National Academy of Engineering},<br \/>\r\nkeywords = {},<br \/>\r\npubstate = {published},<br \/>\r\ntppubtype = {article}<br \/>\r\n}<br \/>\r\n<\/pre><\/div><p class=\"tp_close_menu\"><a class=\"tp_close\" onclick=\"teachpress_pub_showhide('21','tp_bibtex')\">Close<\/a><\/p><\/div><\/div><\/div><h3 class=\"tp_h3\" id=\"tp_h3_2019\">2019<\/h3><div class=\"tp_publication tp_publication_inproceedings\"><div class=\"tp_pub_info\"><p class=\"tp_pub_author\"> Martinez, Victor R;  Somandepalli, Krishna;  Singla, Karan;  Ramakrishna, Anil;  Uhls, Yalda T;  Narayanan, Shrikanth<\/p><p class=\"tp_pub_title\">Violence rating prediction from movie scripts <span class=\"tp_pub_type inproceedings\">Inproceedings<\/span> <\/p><p class=\"tp_pub_additional\"><span class=\"tp_pub_additional_in\">In: <\/span><span class=\"tp_pub_additional_booktitle\">Proceedings of the AAAI Conference on Artificial Intelligence, <\/span><span class=\"tp_pub_additional_pages\">pp. 671\u2013678, <\/span><span class=\"tp_pub_additional_year\">2019<\/span>.<\/p><p class=\"tp_pub_menu\"><span class=\"tp_bibtex_link\"><a id=\"tp_bibtex_sh_12\" class=\"tp_show\" onclick=\"teachpress_pub_showhide('12','tp_bibtex')\" title=\"Show BibTeX entry\" style=\"cursor:pointer;\">BibTeX<\/a><\/span> | <span class=\"tp_pub_tags_label\">Tags: <\/span><\/p><div class=\"tp_bibtex\" id=\"tp_bibtex_12\" style=\"display:none;\"><div class=\"tp_bibtex_entry\"><pre>@inproceedings{martinez2019violence,<br \/>\r\ntitle = {Violence rating prediction from movie scripts},<br \/>\r\nauthor = {Victor R Martinez and Krishna Somandepalli and Karan Singla and Anil Ramakrishna and Yalda T Uhls and Shrikanth Narayanan},<br \/>\r\nyear  = {2019},<br \/>\r\ndate = {2019-01-01},<br \/>\r\nbooktitle = {Proceedings of the AAAI Conference on Artificial Intelligence},<br \/>\r\nvolume = {33},<br \/>\r\nnumber = {01},<br \/>\r\npages = {671--678},<br \/>\r\nkeywords = {},<br \/>\r\npubstate = {published},<br \/>\r\ntppubtype = {inproceedings}<br \/>\r\n}<br \/>\r\n<\/pre><\/div><p class=\"tp_close_menu\"><a class=\"tp_close\" onclick=\"teachpress_pub_showhide('12','tp_bibtex')\">Close<\/a><\/p><\/div><\/div><\/div><div class=\"tp_publication tp_publication_inproceedings\"><div class=\"tp_pub_info\"><p class=\"tp_pub_author\"> Sharma, Rahul;  Somandepalli, Krishna;  Narayanan, Shrikanth<\/p><p class=\"tp_pub_title\">Toward visual voice activity detection for unconstrained videos <span class=\"tp_pub_type inproceedings\">Inproceedings<\/span> <\/p><p class=\"tp_pub_additional\"><span class=\"tp_pub_additional_in\">In: <\/span><span class=\"tp_pub_additional_booktitle\">2019 IEEE International Conference on Image Processing (ICIP), <\/span><span class=\"tp_pub_additional_pages\">pp. 2991\u20132995, <\/span><span class=\"tp_pub_additional_organization\">IEEE <\/span><span class=\"tp_pub_additional_year\">2019<\/span>.<\/p><p class=\"tp_pub_menu\"><span class=\"tp_bibtex_link\"><a id=\"tp_bibtex_sh_11\" class=\"tp_show\" onclick=\"teachpress_pub_showhide('11','tp_bibtex')\" title=\"Show BibTeX entry\" style=\"cursor:pointer;\">BibTeX<\/a><\/span> | <span class=\"tp_pub_tags_label\">Tags: <\/span><\/p><div class=\"tp_bibtex\" id=\"tp_bibtex_11\" style=\"display:none;\"><div class=\"tp_bibtex_entry\"><pre>@inproceedings{sharma2019toward,<br \/>\r\ntitle = {Toward visual voice activity detection for unconstrained videos},<br \/>\r\nauthor = {Rahul Sharma and Krishna Somandepalli and Shrikanth Narayanan},<br \/>\r\nyear  = {2019},<br \/>\r\ndate = {2019-01-01},<br \/>\r\nbooktitle = {2019 IEEE International Conference on Image Processing (ICIP)},<br \/>\r\npages = {2991--2995},<br \/>\r\norganization = {IEEE},<br \/>\r\nkeywords = {},<br \/>\r\npubstate = {published},<br \/>\r\ntppubtype = {inproceedings}<br \/>\r\n}<br \/>\r\n<\/pre><\/div><p class=\"tp_close_menu\"><a class=\"tp_close\" onclick=\"teachpress_pub_showhide('11','tp_bibtex')\">Close<\/a><\/p><\/div><\/div><\/div><div class=\"tp_publication tp_publication_inproceedings\"><div class=\"tp_pub_info\"><p class=\"tp_pub_author\"> Hebbar, Rajat;  Somandepalli, Krishna;  Narayanan, Shrikanth<\/p><p class=\"tp_pub_title\">Robust speech activity detection in movie audio: Data resources and experimental evaluation <span class=\"tp_pub_type inproceedings\">Inproceedings<\/span> <\/p><p class=\"tp_pub_additional\"><span class=\"tp_pub_additional_in\">In: <\/span><span class=\"tp_pub_additional_booktitle\">ICASSP 2019-2019 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP), <\/span><span class=\"tp_pub_additional_pages\">pp. 4105\u20134109, <\/span><span class=\"tp_pub_additional_organization\">IEEE <\/span><span class=\"tp_pub_additional_year\">2019<\/span>.<\/p><p class=\"tp_pub_menu\"><span class=\"tp_bibtex_link\"><a id=\"tp_bibtex_sh_10\" class=\"tp_show\" onclick=\"teachpress_pub_showhide('10','tp_bibtex')\" title=\"Show BibTeX entry\" style=\"cursor:pointer;\">BibTeX<\/a><\/span> | <span class=\"tp_pub_tags_label\">Tags: <\/span><\/p><div class=\"tp_bibtex\" id=\"tp_bibtex_10\" style=\"display:none;\"><div class=\"tp_bibtex_entry\"><pre>@inproceedings{hebbar2019robust,<br \/>\r\ntitle = {Robust speech activity detection in movie audio: Data resources and experimental evaluation},<br \/>\r\nauthor = {Rajat Hebbar and Krishna Somandepalli and Shrikanth Narayanan},<br \/>\r\nyear  = {2019},<br \/>\r\ndate = {2019-01-01},<br \/>\r\nbooktitle = {ICASSP 2019-2019 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP)},<br \/>\r\npages = {4105--4109},<br \/>\r\norganization = {IEEE},<br \/>\r\nkeywords = {},<br \/>\r\npubstate = {published},<br \/>\r\ntppubtype = {inproceedings}<br \/>\r\n}<br \/>\r\n<\/pre><\/div><p class=\"tp_close_menu\"><a class=\"tp_close\" onclick=\"teachpress_pub_showhide('10','tp_bibtex')\">Close<\/a><\/p><\/div><\/div><\/div><div class=\"tp_publication tp_publication_inproceedings\"><div class=\"tp_pub_info\"><p class=\"tp_pub_author\"> Somandepalli, Krishna;  Narayanan, Shrikanth<\/p><p class=\"tp_pub_title\">Reinforcing self-expressive representation with constraint propagation for face clustering in movies <span class=\"tp_pub_type inproceedings\">Inproceedings<\/span> <\/p><p class=\"tp_pub_additional\"><span class=\"tp_pub_additional_in\">In: <\/span><span class=\"tp_pub_additional_booktitle\">ICASSP 2019-2019 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP), <\/span><span class=\"tp_pub_additional_pages\">pp. 4065\u20134069, <\/span><span class=\"tp_pub_additional_organization\">IEEE <\/span><span class=\"tp_pub_additional_year\">2019<\/span>.<\/p><p class=\"tp_pub_menu\"><span class=\"tp_bibtex_link\"><a id=\"tp_bibtex_sh_9\" class=\"tp_show\" onclick=\"teachpress_pub_showhide('9','tp_bibtex')\" title=\"Show BibTeX entry\" style=\"cursor:pointer;\">BibTeX<\/a><\/span> | <span class=\"tp_pub_tags_label\">Tags: <\/span><\/p><div class=\"tp_bibtex\" id=\"tp_bibtex_9\" style=\"display:none;\"><div class=\"tp_bibtex_entry\"><pre>@inproceedings{somandepalli2019reinforcing,<br \/>\r\ntitle = {Reinforcing self-expressive representation with constraint propagation for face clustering in movies},<br \/>\r\nauthor = {Krishna Somandepalli and Shrikanth Narayanan},<br \/>\r\nyear  = {2019},<br \/>\r\ndate = {2019-01-01},<br \/>\r\nbooktitle = {ICASSP 2019-2019 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP)},<br \/>\r\npages = {4065--4069},<br \/>\r\norganization = {IEEE},<br \/>\r\nkeywords = {},<br \/>\r\npubstate = {published},<br \/>\r\ntppubtype = {inproceedings}<br \/>\r\n}<br \/>\r\n<\/pre><\/div><p class=\"tp_close_menu\"><a class=\"tp_close\" onclick=\"teachpress_pub_showhide('9','tp_bibtex')\">Close<\/a><\/p><\/div><\/div><\/div><div class=\"tp_publication tp_publication_article\"><div class=\"tp_pub_info\"><p class=\"tp_pub_author\"> Somandepalli, Krishna;  Kumar, Naveen;  Travadi, Ruchir;  Narayanan, Shrikanth<\/p><p class=\"tp_pub_title\">Multimodal representation learning using deep multiset canonical correlation <span class=\"tp_pub_type article\">Journal Article<\/span> <\/p><p class=\"tp_pub_additional\"><span class=\"tp_pub_additional_in\">In: <\/span><span class=\"tp_pub_additional_journal\">arXiv preprint arXiv:1904.01775, <\/span><span class=\"tp_pub_additional_year\">2019<\/span>.<\/p><p class=\"tp_pub_menu\"><span class=\"tp_bibtex_link\"><a id=\"tp_bibtex_sh_8\" class=\"tp_show\" onclick=\"teachpress_pub_showhide('8','tp_bibtex')\" title=\"Show BibTeX entry\" style=\"cursor:pointer;\">BibTeX<\/a><\/span> | <span class=\"tp_pub_tags_label\">Tags: <\/span><\/p><div class=\"tp_bibtex\" id=\"tp_bibtex_8\" style=\"display:none;\"><div class=\"tp_bibtex_entry\"><pre>@article{somandepalli2019multimodal,<br \/>\r\ntitle = {Multimodal representation learning using deep multiset canonical correlation},<br \/>\r\nauthor = {Krishna Somandepalli and Naveen Kumar and Ruchir Travadi and Shrikanth Narayanan},<br \/>\r\nyear  = {2019},<br \/>\r\ndate = {2019-01-01},<br \/>\r\njournal = {arXiv preprint arXiv:1904.01775},<br \/>\r\nkeywords = {},<br \/>\r\npubstate = {published},<br \/>\r\ntppubtype = {article}<br \/>\r\n}<br \/>\r\n<\/pre><\/div><p class=\"tp_close_menu\"><a class=\"tp_close\" onclick=\"teachpress_pub_showhide('8','tp_bibtex')\">Close<\/a><\/p><\/div><\/div><\/div><h3 class=\"tp_h3\" id=\"tp_h3_2018\">2018<\/h3><div class=\"tp_publication tp_publication_inproceedings\"><div class=\"tp_pub_info\"><p class=\"tp_pub_author\"> Somandepalli, Krishna;  Martinez, Victor;  Kumar, Naveen;  Narayanan, Shrikanth<\/p><p class=\"tp_pub_title\"><a class=\"tp_title_link\" onclick=\"teachpress_pub_showhide('14','tp_links')\" style=\"cursor:pointer;\">Multimodal Representation of Advertisements Using Segment-Level Autoencoders<\/a> <span class=\"tp_pub_type inproceedings\">Inproceedings<\/span> <\/p><p class=\"tp_pub_additional\"><span class=\"tp_pub_additional_in\">In: <\/span><span class=\"tp_pub_additional_booktitle\">Proceedings of the 20th ACM International Conference on Multimodal Interaction, <\/span><span class=\"tp_pub_additional_pages\">pp. 418\u2013422, <\/span><span class=\"tp_pub_additional_publisher\">Association for Computing Machinery, <\/span><span class=\"tp_pub_additional_address\">Boulder, CO, USA, <\/span><span class=\"tp_pub_additional_year\">2018<\/span>, <span class=\"tp_pub_additional_isbn\">ISBN: 9781450356923<\/span>.<\/p><p class=\"tp_pub_menu\"><span class=\"tp_abstract_link\"><a id=\"tp_abstract_sh_14\" class=\"tp_show\" onclick=\"teachpress_pub_showhide('14','tp_abstract')\" title=\"Show abstract\" style=\"cursor:pointer;\">Abstract<\/a><\/span> | <span class=\"tp_resource_link\"><a id=\"tp_links_sh_14\" class=\"tp_show\" onclick=\"teachpress_pub_showhide('14','tp_links')\" title=\"Show links and resources\" style=\"cursor:pointer;\">Links<\/a><\/span> | <span class=\"tp_bibtex_link\"><a id=\"tp_bibtex_sh_14\" class=\"tp_show\" onclick=\"teachpress_pub_showhide('14','tp_bibtex')\" title=\"Show BibTeX entry\" style=\"cursor:pointer;\">BibTeX<\/a><\/span> | <span class=\"tp_pub_tags_label\">Tags: <\/span><a rel=\"nofollow\" href=\"https:\/\/sail.usc.edu:\/ccmi\/publications\/?tgid=3#tppubs\" title=\"Show all publications which have a relationship to this tag\">advertisements<\/a>, <a rel=\"nofollow\" href=\"https:\/\/sail.usc.edu:\/ccmi\/publications\/?tgid=1#tppubs\" title=\"Show all publications which have a relationship to this tag\">autoencoders<\/a>, <a rel=\"nofollow\" href=\"https:\/\/sail.usc.edu:\/ccmi\/publications\/?tgid=2#tppubs\" title=\"Show all publications which have a relationship to this tag\">multimodal joint representation<\/a><\/p><div class=\"tp_bibtex\" id=\"tp_bibtex_14\" style=\"display:none;\"><div class=\"tp_bibtex_entry\"><pre>@inproceedings{10.1145\/3242969.3243026,<br \/>\r\ntitle = {Multimodal Representation of Advertisements Using Segment-Level Autoencoders},<br \/>\r\nauthor = {Krishna Somandepalli and Victor Martinez and Naveen Kumar and Shrikanth Narayanan},<br \/>\r\nurl = {https:\/\/doi.org\/10.1145\/3242969.3243026},<br \/>\r\ndoi = {10.1145\/3242969.3243026},<br \/>\r\nisbn = {9781450356923},<br \/>\r\nyear  = {2018},<br \/>\r\ndate = {2018-01-01},<br \/>\r\nbooktitle = {Proceedings of the 20th ACM International Conference on Multimodal Interaction},<br \/>\r\npages = {418\u2013422},<br \/>\r\npublisher = {Association for Computing Machinery},<br \/>\r\naddress = {Boulder, CO, USA},<br \/>\r\nseries = {ICMI '18},<br \/>\r\nabstract = {Automatic analysis of advertisements (ads) poses an interesting problem for learning <br \/>\r\nmultimodal representations. A promising direction of research is the development of <br \/>\r\ndeep neural network autoencoders to obtain inter-modal and intra-modal representations. <br \/>\r\nIn this work, we propose a system to obtain segment-level unimodal and joint representations. <br \/>\r\nThese features are concatenated, and then averaged across the duration of an ad to <br \/>\r\nobtain a single multimodal representation. The autoencoders are trained using segments <br \/>\r\ngenerated by time-aligning frames between the audio and video modalities with forward <br \/>\r\nand backward context. In order to assess the multimodal representations, we consider <br \/>\r\nthe tasks of classifying an ad as funny or exciting in a publicly available dataset <br \/>\r\nof 2,720 ads. For this purpose we train the segment-level autoencoders on a larger, <br \/>\r\nunlabeled dataset of 9,740 ads, agnostic of the test set. Our experiments show that: <br \/>\r\n1) the multimodal representations outperform joint and unimodal representations, 2) <br \/>\r\nthe different representations we learn are complementary to each other, and 3) the <br \/>\r\nsegment-level multimodal representations perform better than classical autoencoders <br \/>\r\nand cross-modal representations -- within the context of the two classification tasks. <br \/>\r\nWe obtain an improvement of about 5% in classification accuracy compared to a competitive <br \/>\r\nbaseline.},<br \/>\r\nkeywords = {advertisements, autoencoders, multimodal joint representation},<br \/>\r\npubstate = {published},<br \/>\r\ntppubtype = {inproceedings}<br \/>\r\n}<br \/>\r\n<\/pre><\/div><p class=\"tp_close_menu\"><a class=\"tp_close\" onclick=\"teachpress_pub_showhide('14','tp_bibtex')\">Close<\/a><\/p><\/div><div class=\"tp_abstract\" id=\"tp_abstract_14\" style=\"display:none;\"><div class=\"tp_abstract_entry\">Automatic analysis of advertisements (ads) poses an interesting problem for learning <br \/>\r\nmultimodal representations. A promising direction of research is the development of <br \/>\r\ndeep neural network autoencoders to obtain inter-modal and intra-modal representations. <br \/>\r\nIn this work, we propose a system to obtain segment-level unimodal and joint representations. <br \/>\r\nThese features are concatenated, and then averaged across the duration of an ad to <br \/>\r\nobtain a single multimodal representation. The autoencoders are trained using segments <br \/>\r\ngenerated by time-aligning frames between the audio and video modalities with forward <br \/>\r\nand backward context. In order to assess the multimodal representations, we consider <br \/>\r\nthe tasks of classifying an ad as funny or exciting in a publicly available dataset <br \/>\r\nof 2,720 ads. For this purpose we train the segment-level autoencoders on a larger, <br \/>\r\nunlabeled dataset of 9,740 ads, agnostic of the test set. Our experiments show that: <br \/>\r\n1) the multimodal representations outperform joint and unimodal representations, 2) <br \/>\r\nthe different representations we learn are complementary to each other, and 3) the <br \/>\r\nsegment-level multimodal representations perform better than classical autoencoders <br \/>\r\nand cross-modal representations -- within the context of the two classification tasks. <br \/>\r\nWe obtain an improvement of about 5% in classification accuracy compared to a competitive <br \/>\r\nbaseline.<\/div><p class=\"tp_close_menu\"><a class=\"tp_close\" onclick=\"teachpress_pub_showhide('14','tp_abstract')\">Close<\/a><\/p><\/div><div class=\"tp_links\" id=\"tp_links_14\" style=\"display:none;\"><div class=\"tp_links_entry\"><ul class=\"tp_pub_list\"><li><i class=\"fas fa-globe\"><\/i><a class=\"tp_pub_list\" href=\"https:\/\/doi.org\/10.1145\/3242969.3243026\" title=\"https:\/\/doi.org\/10.1145\/3242969.3243026\" target=\"_blank\">https:\/\/doi.org\/10.1145\/3242969.3243026<\/a><\/li><li><i class=\"ai ai-doi\"><\/i><a class=\"tp_pub_list\" href=\"https:\/\/dx.doi.org\/10.1145\/3242969.3243026\" title=\"Follow DOI:10.1145\/3242969.3243026\" target=\"_blank\">doi:10.1145\/3242969.3243026<\/a><\/li><\/ul><\/div><p class=\"tp_close_menu\"><a class=\"tp_close\" onclick=\"teachpress_pub_showhide('14','tp_links')\">Close<\/a><\/p><\/div><\/div><\/div><div class=\"tp_publication tp_publication_inproceedings\"><div class=\"tp_pub_info\"><p class=\"tp_pub_author\"> Hebbar, Rajat;  Somandepalli, Krishna;  Narayanan, Shrikanth<\/p><p class=\"tp_pub_title\"><a class=\"tp_title_link\" onclick=\"teachpress_pub_showhide('15','tp_links')\" style=\"cursor:pointer;\">Improving Gender Identification in Movie Audio Using Cross-Domain Data<\/a> <span class=\"tp_pub_type inproceedings\">Inproceedings<\/span> <\/p><p class=\"tp_pub_additional\"><span class=\"tp_pub_additional_in\">In: <\/span><span class=\"tp_pub_additional_booktitle\">Proc. Interspeech 2018, <\/span><span class=\"tp_pub_additional_pages\">pp. 282\u2013286, <\/span><span class=\"tp_pub_additional_year\">2018<\/span>.<\/p><p class=\"tp_pub_menu\"><span class=\"tp_resource_link\"><a id=\"tp_links_sh_15\" class=\"tp_show\" onclick=\"teachpress_pub_showhide('15','tp_links')\" title=\"Show links and resources\" style=\"cursor:pointer;\">Links<\/a><\/span> | <span class=\"tp_bibtex_link\"><a id=\"tp_bibtex_sh_15\" class=\"tp_show\" onclick=\"teachpress_pub_showhide('15','tp_bibtex')\" title=\"Show BibTeX entry\" style=\"cursor:pointer;\">BibTeX<\/a><\/span> | <span class=\"tp_pub_tags_label\">Tags: <\/span><\/p><div class=\"tp_bibtex\" id=\"tp_bibtex_15\" style=\"display:none;\"><div class=\"tp_bibtex_entry\"><pre>@inproceedings{hebbar18_interspeech,<br \/>\r\ntitle = {Improving Gender Identification in Movie Audio Using Cross-Domain Data},<br \/>\r\nauthor = {Rajat Hebbar and Krishna Somandepalli and Shrikanth Narayanan},<br \/>\r\ndoi = {10.21437\/Interspeech.2018-1462},<br \/>\r\nyear  = {2018},<br \/>\r\ndate = {2018-01-01},<br \/>\r\nbooktitle = {Proc. Interspeech 2018},<br \/>\r\npages = {282--286},<br \/>\r\nkeywords = {},<br \/>\r\npubstate = {published},<br \/>\r\ntppubtype = {inproceedings}<br \/>\r\n}<br \/>\r\n<\/pre><\/div><p class=\"tp_close_menu\"><a class=\"tp_close\" onclick=\"teachpress_pub_showhide('15','tp_bibtex')\">Close<\/a><\/p><\/div><div class=\"tp_links\" id=\"tp_links_15\" style=\"display:none;\"><div class=\"tp_links_entry\"><ul class=\"tp_pub_list\"><li><i class=\"ai ai-doi\"><\/i><a class=\"tp_pub_list\" href=\"https:\/\/dx.doi.org\/10.21437\/Interspeech.2018-1462\" title=\"Follow DOI:10.21437\/Interspeech.2018-1462\" target=\"_blank\">doi:10.21437\/Interspeech.2018-1462<\/a><\/li><\/ul><\/div><p class=\"tp_close_menu\"><a class=\"tp_close\" onclick=\"teachpress_pub_showhide('15','tp_links')\">Close<\/a><\/p><\/div><\/div><\/div><div class=\"tp_publication tp_publication_article\"><div class=\"tp_pub_info\"><p class=\"tp_pub_author\"> Somandepalli, Krishna;  Kumar, Naveen;  Guha, Tanaya;  Narayanan, Shrikanth S<\/p><p class=\"tp_pub_title\"><a class=\"tp_title_link\" onclick=\"teachpress_pub_showhide('16','tp_links')\" style=\"cursor:pointer;\">Unsupervised Discovery of Character Dictionaries in Animation Movies<\/a> <span class=\"tp_pub_type article\">Journal Article<\/span> <\/p><p class=\"tp_pub_additional\"><span class=\"tp_pub_additional_in\">In: <\/span><span class=\"tp_pub_additional_journal\">IEEE Transactions on Multimedia, <\/span><span class=\"tp_pub_additional_volume\">20 <\/span><span class=\"tp_pub_additional_number\">(3), <\/span><span class=\"tp_pub_additional_pages\">pp. 539-551, <\/span><span class=\"tp_pub_additional_year\">2018<\/span>.<\/p><p class=\"tp_pub_menu\"><span class=\"tp_resource_link\"><a id=\"tp_links_sh_16\" class=\"tp_show\" onclick=\"teachpress_pub_showhide('16','tp_links')\" title=\"Show links and resources\" style=\"cursor:pointer;\">Links<\/a><\/span> | <span class=\"tp_bibtex_link\"><a id=\"tp_bibtex_sh_16\" class=\"tp_show\" onclick=\"teachpress_pub_showhide('16','tp_bibtex')\" title=\"Show BibTeX entry\" style=\"cursor:pointer;\">BibTeX<\/a><\/span> | <span class=\"tp_pub_tags_label\">Tags: <\/span><\/p><div class=\"tp_bibtex\" id=\"tp_bibtex_16\" style=\"display:none;\"><div class=\"tp_bibtex_entry\"><pre>@article{8017484,<br \/>\r\ntitle = {Unsupervised Discovery of Character Dictionaries in Animation Movies},<br \/>\r\nauthor = {Krishna Somandepalli and Naveen Kumar and Tanaya Guha and Shrikanth S Narayanan},<br \/>\r\ndoi = {10.1109\/TMM.2017.2745712},<br \/>\r\nyear  = {2018},<br \/>\r\ndate = {2018-01-01},<br \/>\r\njournal = {IEEE Transactions on Multimedia},<br \/>\r\nvolume = {20},<br \/>\r\nnumber = {3},<br \/>\r\npages = {539-551},<br \/>\r\nkeywords = {},<br \/>\r\npubstate = {published},<br \/>\r\ntppubtype = {article}<br \/>\r\n}<br \/>\r\n<\/pre><\/div><p class=\"tp_close_menu\"><a class=\"tp_close\" onclick=\"teachpress_pub_showhide('16','tp_bibtex')\">Close<\/a><\/p><\/div><div class=\"tp_links\" id=\"tp_links_16\" style=\"display:none;\"><div class=\"tp_links_entry\"><ul class=\"tp_pub_list\"><li><i class=\"ai ai-doi\"><\/i><a class=\"tp_pub_list\" href=\"https:\/\/dx.doi.org\/10.1109\/TMM.2017.2745712\" title=\"Follow DOI:10.1109\/TMM.2017.2745712\" target=\"_blank\">doi:10.1109\/TMM.2017.2745712<\/a><\/li><\/ul><\/div><p class=\"tp_close_menu\"><a class=\"tp_close\" onclick=\"teachpress_pub_showhide('16','tp_links')\">Close<\/a><\/p><\/div><\/div><\/div><h3 class=\"tp_h3\" id=\"tp_h3_2017\">2017<\/h3><div class=\"tp_publication tp_publication_inproceedings\"><div class=\"tp_pub_info\"><p class=\"tp_pub_author\"> Ramakrishna, Anil;  Martinez, Victor R;  Malandrakis, Nikolaos;  Singla, Karan;  Narayanan, Shrikanth<\/p><p class=\"tp_pub_title\"><a class=\"tp_title_link\" onclick=\"teachpress_pub_showhide('17','tp_links')\" style=\"cursor:pointer;\">Linguistic analysis of differences in portrayal of movie characters<\/a> <span class=\"tp_pub_type inproceedings\">Inproceedings<\/span> <\/p><p class=\"tp_pub_additional\"><span class=\"tp_pub_additional_in\">In: <\/span><span class=\"tp_pub_additional_booktitle\">Proceedings of the 55th Annual Meeting of the Association for Computational Linguistics (Volume 1: Long Papers), <\/span><span class=\"tp_pub_additional_pages\">pp. 1669\u20131678, <\/span><span class=\"tp_pub_additional_address\">Vancouver, Canada, <\/span><span class=\"tp_pub_additional_year\">2017<\/span>.<\/p><p class=\"tp_pub_menu\"><span class=\"tp_abstract_link\"><a id=\"tp_abstract_sh_17\" class=\"tp_show\" onclick=\"teachpress_pub_showhide('17','tp_abstract')\" title=\"Show abstract\" style=\"cursor:pointer;\">Abstract<\/a><\/span> | <span class=\"tp_resource_link\"><a id=\"tp_links_sh_17\" class=\"tp_show\" onclick=\"teachpress_pub_showhide('17','tp_links')\" title=\"Show links and resources\" style=\"cursor:pointer;\">Links<\/a><\/span> | <span class=\"tp_bibtex_link\"><a id=\"tp_bibtex_sh_17\" class=\"tp_show\" onclick=\"teachpress_pub_showhide('17','tp_bibtex')\" title=\"Show BibTeX entry\" style=\"cursor:pointer;\">BibTeX<\/a><\/span> | <span class=\"tp_pub_tags_label\">Tags: <\/span><\/p><div class=\"tp_bibtex\" id=\"tp_bibtex_17\" style=\"display:none;\"><div class=\"tp_bibtex_entry\"><pre>@inproceedings{ramakrishna-etal-2017-linguistic,<br \/>\r\ntitle = {Linguistic analysis of differences in portrayal of movie characters},<br \/>\r\nauthor = {Anil Ramakrishna and Victor R Martinez and Nikolaos Malandrakis and Karan Singla and Shrikanth Narayanan},<br \/>\r\nurl = {https:\/\/aclanthology.org\/P17-1153},<br \/>\r\ndoi = {10.18653\/v1\/P17-1153},<br \/>\r\nyear  = {2017},<br \/>\r\ndate = {2017-01-01},<br \/>\r\nurldate = {2017-01-01},<br \/>\r\nbooktitle = {Proceedings of the 55th Annual Meeting of the Association for Computational Linguistics (Volume 1: Long Papers)},<br \/>\r\npages = {1669--1678},<br \/>\r\naddress = {Vancouver, Canada},<br \/>\r\nabstract = {We examine differences in portrayal of characters in movies using psycholinguistic and graph theoretic measures computed directly from screenplays. Differences are examined with respect to characters' gender, race, age and other metadata. Psycholinguistic metrics are extrapolated to dialogues in movies using a linear regression model built on a set of manually annotated seed words. Interesting patterns are revealed about relationships between genders of production team and the gender ratio of characters. Several correlations are noted between gender, race, age of characters and the linguistic metrics.},<br \/>\r\nkeywords = {},<br \/>\r\npubstate = {published},<br \/>\r\ntppubtype = {inproceedings}<br \/>\r\n}<br \/>\r\n<\/pre><\/div><p class=\"tp_close_menu\"><a class=\"tp_close\" onclick=\"teachpress_pub_showhide('17','tp_bibtex')\">Close<\/a><\/p><\/div><div class=\"tp_abstract\" id=\"tp_abstract_17\" style=\"display:none;\"><div class=\"tp_abstract_entry\">We examine differences in portrayal of characters in movies using psycholinguistic and graph theoretic measures computed directly from screenplays. Differences are examined with respect to characters&#039; gender, race, age and other metadata. Psycholinguistic metrics are extrapolated to dialogues in movies using a linear regression model built on a set of manually annotated seed words. Interesting patterns are revealed about relationships between genders of production team and the gender ratio of characters. Several correlations are noted between gender, race, age of characters and the linguistic metrics.<\/div><p class=\"tp_close_menu\"><a class=\"tp_close\" onclick=\"teachpress_pub_showhide('17','tp_abstract')\">Close<\/a><\/p><\/div><div class=\"tp_links\" id=\"tp_links_17\" style=\"display:none;\"><div class=\"tp_links_entry\"><ul class=\"tp_pub_list\"><li><i class=\"fas fa-globe\"><\/i><a class=\"tp_pub_list\" href=\"https:\/\/aclanthology.org\/P17-1153\" title=\"https:\/\/aclanthology.org\/P17-1153\" target=\"_blank\">https:\/\/aclanthology.org\/P17-1153<\/a><\/li><li><i class=\"ai ai-doi\"><\/i><a class=\"tp_pub_list\" href=\"https:\/\/dx.doi.org\/10.18653\/v1\/P17-1153\" title=\"Follow DOI:10.18653\/v1\/P17-1153\" target=\"_blank\">doi:10.18653\/v1\/P17-1153<\/a><\/li><\/ul><\/div><p class=\"tp_close_menu\"><a class=\"tp_close\" onclick=\"teachpress_pub_showhide('17','tp_links')\">Close<\/a><\/p><\/div><\/div><\/div><h3 class=\"tp_h3\" id=\"tp_h3_2016\">2016<\/h3><div class=\"tp_publication tp_publication_inproceedings\"><div class=\"tp_pub_info\"><p class=\"tp_pub_author\"> Tadimari, Adarsh;  Kumar, Naveen;  Guha, Tanaya;  Narayanan, Shrikanth S<\/p><p class=\"tp_pub_title\"><a class=\"tp_title_link\" onclick=\"teachpress_pub_showhide('25','tp_links')\" style=\"cursor:pointer;\">Opening big in box office? Trailer content can help<\/a> <span class=\"tp_pub_type inproceedings\">Inproceedings<\/span> <\/p><p class=\"tp_pub_additional\"><span class=\"tp_pub_additional_in\">In: <\/span><span class=\"tp_pub_additional_booktitle\">2016 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP), <\/span><span class=\"tp_pub_additional_pages\">pp. 2777-2781, <\/span><span class=\"tp_pub_additional_year\">2016<\/span>.<\/p><p class=\"tp_pub_menu\"><span class=\"tp_resource_link\"><a id=\"tp_links_sh_25\" class=\"tp_show\" onclick=\"teachpress_pub_showhide('25','tp_links')\" title=\"Show links and resources\" style=\"cursor:pointer;\">Links<\/a><\/span> | <span class=\"tp_bibtex_link\"><a id=\"tp_bibtex_sh_25\" class=\"tp_show\" onclick=\"teachpress_pub_showhide('25','tp_bibtex')\" title=\"Show BibTeX entry\" style=\"cursor:pointer;\">BibTeX<\/a><\/span> | <span class=\"tp_pub_tags_label\">Tags: <\/span><\/p><div class=\"tp_bibtex\" id=\"tp_bibtex_25\" style=\"display:none;\"><div class=\"tp_bibtex_entry\"><pre>@inproceedings{7472183,<br \/>\r\ntitle = {Opening big in box office? Trailer content can help},<br \/>\r\nauthor = {Adarsh Tadimari and Naveen Kumar and Tanaya Guha and Shrikanth S Narayanan},<br \/>\r\ndoi = {10.1109\/ICASSP.2016.7472183},<br \/>\r\nyear  = {2016},<br \/>\r\ndate = {2016-01-01},<br \/>\r\nbooktitle = {2016 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP)},<br \/>\r\npages = {2777-2781},<br \/>\r\nkeywords = {},<br \/>\r\npubstate = {published},<br \/>\r\ntppubtype = {inproceedings}<br \/>\r\n}<br \/>\r\n<\/pre><\/div><p class=\"tp_close_menu\"><a class=\"tp_close\" onclick=\"teachpress_pub_showhide('25','tp_bibtex')\">Close<\/a><\/p><\/div><div class=\"tp_links\" id=\"tp_links_25\" style=\"display:none;\"><div class=\"tp_links_entry\"><ul class=\"tp_pub_list\"><li><i class=\"ai ai-doi\"><\/i><a class=\"tp_pub_list\" href=\"https:\/\/dx.doi.org\/10.1109\/ICASSP.2016.7472183\" title=\"Follow DOI:10.1109\/ICASSP.2016.7472183\" target=\"_blank\">doi:10.1109\/ICASSP.2016.7472183<\/a><\/li><\/ul><\/div><p class=\"tp_close_menu\"><a class=\"tp_close\" onclick=\"teachpress_pub_showhide('25','tp_links')\">Close<\/a><\/p><\/div><\/div><\/div><div class=\"tp_publication tp_publication_inproceedings\"><div class=\"tp_pub_info\"><p class=\"tp_pub_author\"> Nasir, Md;  Kumar, Naveen;  Georgiou, Panayiotis;  Narayanan, Shrikanth S<\/p><p class=\"tp_pub_title\"><a class=\"tp_title_link\" onclick=\"teachpress_pub_showhide('24','tp_links')\" style=\"cursor:pointer;\">Robust Multichannel Gender Classification from Speech in Movie Audio<\/a> <span class=\"tp_pub_type inproceedings\">Inproceedings<\/span> <\/p><p class=\"tp_pub_additional\"><span class=\"tp_pub_additional_in\">In: <\/span><span class=\"tp_pub_additional_booktitle\">Proceedings of Interspeech, <\/span><span class=\"tp_pub_additional_year\">2016<\/span>.<\/p><p class=\"tp_pub_menu\"><span class=\"tp_resource_link\"><a id=\"tp_links_sh_24\" class=\"tp_show\" onclick=\"teachpress_pub_showhide('24','tp_links')\" title=\"Show links and resources\" style=\"cursor:pointer;\">Links<\/a><\/span> | <span class=\"tp_bibtex_link\"><a id=\"tp_bibtex_sh_24\" class=\"tp_show\" onclick=\"teachpress_pub_showhide('24','tp_bibtex')\" title=\"Show BibTeX entry\" style=\"cursor:pointer;\">BibTeX<\/a><\/span> | <span class=\"tp_pub_tags_label\">Tags: <\/span><\/p><div class=\"tp_bibtex\" id=\"tp_bibtex_24\" style=\"display:none;\"><div class=\"tp_bibtex_entry\"><pre>@inproceedings{Nasir2016RobustMultichannelGenderClassification,<br \/>\r\ntitle = {Robust Multichannel Gender Classification from Speech in Movie Audio},<br \/>\r\nauthor = {Md Nasir and Naveen Kumar and Panayiotis Georgiou and Shrikanth S Narayanan},<br \/>\r\ndoi = {10.1109\/TCE.2010.5606301},<br \/>\r\nyear  = {2016},<br \/>\r\ndate = {2016-01-01},<br \/>\r\nbooktitle = {Proceedings of Interspeech},<br \/>\r\nkeywords = {},<br \/>\r\npubstate = {published},<br \/>\r\ntppubtype = {inproceedings}<br \/>\r\n}<br \/>\r\n<\/pre><\/div><p class=\"tp_close_menu\"><a class=\"tp_close\" onclick=\"teachpress_pub_showhide('24','tp_bibtex')\">Close<\/a><\/p><\/div><div class=\"tp_links\" id=\"tp_links_24\" style=\"display:none;\"><div class=\"tp_links_entry\"><ul class=\"tp_pub_list\"><li><i class=\"ai ai-doi\"><\/i><a class=\"tp_pub_list\" href=\"https:\/\/dx.doi.org\/10.1109\/TCE.2010.5606301\" title=\"Follow DOI:10.1109\/TCE.2010.5606301\" target=\"_blank\">doi:10.1109\/TCE.2010.5606301<\/a><\/li><\/ul><\/div><p class=\"tp_close_menu\"><a class=\"tp_close\" onclick=\"teachpress_pub_showhide('24','tp_links')\">Close<\/a><\/p><\/div><\/div><\/div><div class=\"tp_publication tp_publication_inproceedings\"><div class=\"tp_pub_info\"><p class=\"tp_pub_author\"> Goyal, Ankit;  Kumar, Naveen;  Guha, Tanaya;  Narayanan, Shrikanth S<\/p><p class=\"tp_pub_title\"><a class=\"tp_title_link\" onclick=\"teachpress_pub_showhide('23','tp_links')\" style=\"cursor:pointer;\">A multimodal mixture-of-experts model for dynamic emotion prediction in movies<\/a> <span class=\"tp_pub_type inproceedings\">Inproceedings<\/span> <\/p><p class=\"tp_pub_additional\"><span class=\"tp_pub_additional_in\">In: <\/span><span class=\"tp_pub_additional_booktitle\">2016 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP), <\/span><span class=\"tp_pub_additional_pages\">pp. 2822-2826, <\/span><span class=\"tp_pub_additional_year\">2016<\/span>.<\/p><p class=\"tp_pub_menu\"><span class=\"tp_resource_link\"><a id=\"tp_links_sh_23\" class=\"tp_show\" onclick=\"teachpress_pub_showhide('23','tp_links')\" title=\"Show links and resources\" style=\"cursor:pointer;\">Links<\/a><\/span> | <span class=\"tp_bibtex_link\"><a id=\"tp_bibtex_sh_23\" class=\"tp_show\" onclick=\"teachpress_pub_showhide('23','tp_bibtex')\" title=\"Show BibTeX entry\" style=\"cursor:pointer;\">BibTeX<\/a><\/span> | <span class=\"tp_pub_tags_label\">Tags: <\/span><\/p><div class=\"tp_bibtex\" id=\"tp_bibtex_23\" style=\"display:none;\"><div class=\"tp_bibtex_entry\"><pre>@inproceedings{7472192,<br \/>\r\ntitle = {A multimodal mixture-of-experts model for dynamic emotion prediction in movies},<br \/>\r\nauthor = {Ankit Goyal and Naveen Kumar and Tanaya Guha and Shrikanth S Narayanan},<br \/>\r\ndoi = {10.1109\/ICASSP.2016.7472192},<br \/>\r\nyear  = {2016},<br \/>\r\ndate = {2016-01-01},<br \/>\r\nbooktitle = {2016 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP)},<br \/>\r\npages = {2822-2826},<br \/>\r\nkeywords = {},<br \/>\r\npubstate = {published},<br \/>\r\ntppubtype = {inproceedings}<br \/>\r\n}<br \/>\r\n<\/pre><\/div><p class=\"tp_close_menu\"><a class=\"tp_close\" onclick=\"teachpress_pub_showhide('23','tp_bibtex')\">Close<\/a><\/p><\/div><div class=\"tp_links\" id=\"tp_links_23\" style=\"display:none;\"><div class=\"tp_links_entry\"><ul class=\"tp_pub_list\"><li><i class=\"ai ai-doi\"><\/i><a class=\"tp_pub_list\" href=\"https:\/\/dx.doi.org\/10.1109\/ICASSP.2016.7472192\" title=\"Follow DOI:10.1109\/ICASSP.2016.7472192\" target=\"_blank\">doi:10.1109\/ICASSP.2016.7472192<\/a><\/li><\/ul><\/div><p class=\"tp_close_menu\"><a class=\"tp_close\" onclick=\"teachpress_pub_showhide('23','tp_links')\">Close<\/a><\/p><\/div><\/div><\/div><h3 class=\"tp_h3\" id=\"tp_h3_2015\">2015<\/h3><div class=\"tp_publication tp_publication_inproceedings\"><div class=\"tp_pub_info\"><p class=\"tp_pub_author\"> Guha, Tanaya;  Kumar, Naveen;  Narayanan, Shrikanth S;  Smith, Stacy L<\/p><p class=\"tp_pub_title\"><a class=\"tp_title_link\" onclick=\"teachpress_pub_showhide('28','tp_links')\" style=\"cursor:pointer;\">Computationally deconstructing movie narratives: An informatics approach<\/a> <span class=\"tp_pub_type inproceedings\">Inproceedings<\/span> <\/p><p class=\"tp_pub_additional\"><span class=\"tp_pub_additional_in\">In: <\/span><span class=\"tp_pub_additional_booktitle\">2015 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP), <\/span><span class=\"tp_pub_additional_pages\">pp. 2264-2268, <\/span><span class=\"tp_pub_additional_year\">2015<\/span>.<\/p><p class=\"tp_pub_menu\"><span class=\"tp_resource_link\"><a id=\"tp_links_sh_28\" class=\"tp_show\" onclick=\"teachpress_pub_showhide('28','tp_links')\" title=\"Show links and resources\" style=\"cursor:pointer;\">Links<\/a><\/span> | <span class=\"tp_bibtex_link\"><a id=\"tp_bibtex_sh_28\" class=\"tp_show\" onclick=\"teachpress_pub_showhide('28','tp_bibtex')\" title=\"Show BibTeX entry\" style=\"cursor:pointer;\">BibTeX<\/a><\/span> | <span class=\"tp_pub_tags_label\">Tags: <\/span><\/p><div class=\"tp_bibtex\" id=\"tp_bibtex_28\" style=\"display:none;\"><div class=\"tp_bibtex_entry\"><pre>@inproceedings{7178374,<br \/>\r\ntitle = {Computationally deconstructing movie narratives: An informatics approach},<br \/>\r\nauthor = {Tanaya Guha and Naveen Kumar and Shrikanth S Narayanan and Stacy L Smith},<br \/>\r\ndoi = {10.1109\/ICASSP.2015.7178374},<br \/>\r\nyear  = {2015},<br \/>\r\ndate = {2015-01-01},<br \/>\r\nbooktitle = {2015 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP)},<br \/>\r\npages = {2264-2268},<br \/>\r\nkeywords = {},<br \/>\r\npubstate = {published},<br \/>\r\ntppubtype = {inproceedings}<br \/>\r\n}<br \/>\r\n<\/pre><\/div><p class=\"tp_close_menu\"><a class=\"tp_close\" onclick=\"teachpress_pub_showhide('28','tp_bibtex')\">Close<\/a><\/p><\/div><div class=\"tp_links\" id=\"tp_links_28\" style=\"display:none;\"><div class=\"tp_links_entry\"><ul class=\"tp_pub_list\"><li><i class=\"ai ai-doi\"><\/i><a class=\"tp_pub_list\" href=\"https:\/\/dx.doi.org\/10.1109\/ICASSP.2015.7178374\" title=\"Follow DOI:10.1109\/ICASSP.2015.7178374\" target=\"_blank\">doi:10.1109\/ICASSP.2015.7178374<\/a><\/li><\/ul><\/div><p class=\"tp_close_menu\"><a class=\"tp_close\" onclick=\"teachpress_pub_showhide('28','tp_links')\">Close<\/a><\/p><\/div><\/div><\/div><div class=\"tp_publication tp_publication_inproceedings\"><div class=\"tp_pub_info\"><p class=\"tp_pub_author\"> Guha, Tanaya;  Huang, Che-Wei;  Kumar, Naveen;  Zhu, Yan;  Narayanan, Shrikanth S<\/p><p class=\"tp_pub_title\"><a class=\"tp_title_link\" onclick=\"teachpress_pub_showhide('29','tp_links')\" style=\"cursor:pointer;\">Gender Representation in Cinematic Content: A Multimodal Approach<\/a> <span class=\"tp_pub_type inproceedings\">Inproceedings<\/span> <\/p><p class=\"tp_pub_additional\"><span class=\"tp_pub_additional_in\">In: <\/span><span class=\"tp_pub_additional_booktitle\">Proceedings of the 2015 ACM on International Conference on Multimodal Interaction, <\/span><span class=\"tp_pub_additional_pages\">pp. 31\u201334, <\/span><span class=\"tp_pub_additional_publisher\">Association for Computing Machinery, <\/span><span class=\"tp_pub_additional_address\">Seattle, Washington, USA, <\/span><span class=\"tp_pub_additional_year\">2015<\/span>, <span class=\"tp_pub_additional_isbn\">ISBN: 9781450339124<\/span>.<\/p><p class=\"tp_pub_menu\"><span class=\"tp_abstract_link\"><a id=\"tp_abstract_sh_29\" class=\"tp_show\" onclick=\"teachpress_pub_showhide('29','tp_abstract')\" title=\"Show abstract\" style=\"cursor:pointer;\">Abstract<\/a><\/span> | <span class=\"tp_resource_link\"><a id=\"tp_links_sh_29\" class=\"tp_show\" onclick=\"teachpress_pub_showhide('29','tp_links')\" title=\"Show links and resources\" style=\"cursor:pointer;\">Links<\/a><\/span> | <span class=\"tp_bibtex_link\"><a id=\"tp_bibtex_sh_29\" class=\"tp_show\" onclick=\"teachpress_pub_showhide('29','tp_bibtex')\" title=\"Show BibTeX entry\" style=\"cursor:pointer;\">BibTeX<\/a><\/span> | <span class=\"tp_pub_tags_label\">Tags: <\/span><a rel=\"nofollow\" href=\"https:\/\/sail.usc.edu:\/ccmi\/publications\/?tgid=6#tppubs\" title=\"Show all publications which have a relationship to this tag\">content analysis<\/a>, <a rel=\"nofollow\" href=\"https:\/\/sail.usc.edu:\/ccmi\/publications\/?tgid=5#tppubs\" title=\"Show all publications which have a relationship to this tag\">gender representation<\/a>, <a rel=\"nofollow\" href=\"https:\/\/sail.usc.edu:\/ccmi\/publications\/?tgid=8#tppubs\" title=\"Show all publications which have a relationship to this tag\">movie<\/a>, <a rel=\"nofollow\" href=\"https:\/\/sail.usc.edu:\/ccmi\/publications\/?tgid=7#tppubs\" title=\"Show all publications which have a relationship to this tag\">multimodal<\/a><\/p><div class=\"tp_bibtex\" id=\"tp_bibtex_29\" style=\"display:none;\"><div class=\"tp_bibtex_entry\"><pre>@inproceedings{10.1145\/2818346.2820778,<br \/>\r\ntitle = {Gender Representation in Cinematic Content: A Multimodal Approach},<br \/>\r\nauthor = {Tanaya Guha and Che-Wei Huang and Naveen Kumar and Yan Zhu and Shrikanth S Narayanan},<br \/>\r\nurl = {https:\/\/doi.org\/10.1145\/2818346.2820778},<br \/>\r\ndoi = {10.1145\/2818346.2820778},<br \/>\r\nisbn = {9781450339124},<br \/>\r\nyear  = {2015},<br \/>\r\ndate = {2015-01-01},<br \/>\r\nbooktitle = {Proceedings of the 2015 ACM on International Conference on Multimodal Interaction},<br \/>\r\npages = {31\u201334},<br \/>\r\npublisher = {Association for Computing Machinery},<br \/>\r\naddress = {Seattle, Washington, USA},<br \/>\r\nseries = {ICMI '15},<br \/>\r\nabstract = {The goal of this paper is to enable an objective understanding of gender portrayals <br \/>\r\nin popular films and media through multimodal content analysis. An automated system <br \/>\r\nfor analyzing gender representation in terms of screen presence and speaking time <br \/>\r\nis developed. First, we perform independent processing of the video and the audio <br \/>\r\ncontent to estimate gender distribution of screen presence at shot level, and of speech <br \/>\r\nat utterance level. A measure of the movie's excitement or intensity is computed using <br \/>\r\naudiovisual features for every scene. This measure is used as a weighting function <br \/>\r\nto combine the gender-based screen\/speaking time information at shot\/utterance level <br \/>\r\nto compute gender representation for the entire movie. Detailed results and analyses <br \/>\r\nare presented on seventeen full length Hollywood movies.},<br \/>\r\nkeywords = {content analysis, gender representation, movie, multimodal},<br \/>\r\npubstate = {published},<br \/>\r\ntppubtype = {inproceedings}<br \/>\r\n}<br \/>\r\n<\/pre><\/div><p class=\"tp_close_menu\"><a class=\"tp_close\" onclick=\"teachpress_pub_showhide('29','tp_bibtex')\">Close<\/a><\/p><\/div><div class=\"tp_abstract\" id=\"tp_abstract_29\" style=\"display:none;\"><div class=\"tp_abstract_entry\">The goal of this paper is to enable an objective understanding of gender portrayals <br \/>\r\nin popular films and media through multimodal content analysis. An automated system <br \/>\r\nfor analyzing gender representation in terms of screen presence and speaking time <br \/>\r\nis developed. First, we perform independent processing of the video and the audio <br \/>\r\ncontent to estimate gender distribution of screen presence at shot level, and of speech <br \/>\r\nat utterance level. A measure of the movie's excitement or intensity is computed using <br \/>\r\naudiovisual features for every scene. This measure is used as a weighting function <br \/>\r\nto combine the gender-based screen\/speaking time information at shot\/utterance level <br \/>\r\nto compute gender representation for the entire movie. Detailed results and analyses <br \/>\r\nare presented on seventeen full length Hollywood movies.<\/div><p class=\"tp_close_menu\"><a class=\"tp_close\" onclick=\"teachpress_pub_showhide('29','tp_abstract')\">Close<\/a><\/p><\/div><div class=\"tp_links\" id=\"tp_links_29\" style=\"display:none;\"><div class=\"tp_links_entry\"><ul class=\"tp_pub_list\"><li><i class=\"fas fa-globe\"><\/i><a class=\"tp_pub_list\" href=\"https:\/\/doi.org\/10.1145\/2818346.2820778\" title=\"https:\/\/doi.org\/10.1145\/2818346.2820778\" target=\"_blank\">https:\/\/doi.org\/10.1145\/2818346.2820778<\/a><\/li><li><i class=\"ai ai-doi\"><\/i><a class=\"tp_pub_list\" href=\"https:\/\/dx.doi.org\/10.1145\/2818346.2820778\" title=\"Follow DOI:10.1145\/2818346.2820778\" target=\"_blank\">doi:10.1145\/2818346.2820778<\/a><\/li><\/ul><\/div><p class=\"tp_close_menu\"><a class=\"tp_close\" onclick=\"teachpress_pub_showhide('29','tp_links')\">Close<\/a><\/p><\/div><\/div><\/div><\/div><\/div>\n","protected":false},"excerpt":{"rendered":"","protected":false},"author":1,"featured_media":0,"parent":0,"menu_order":0,"comment_status":"closed","ping_status":"closed","template":"","meta":{"inline_featured_image":false,"footnotes":""},"acf":[],"_links":{"self":[{"href":"https:\/\/sail.usc.edu:\/ccmi\/wp-json\/wp\/v2\/pages\/54"}],"collection":[{"href":"https:\/\/sail.usc.edu:\/ccmi\/wp-json\/wp\/v2\/pages"}],"about":[{"href":"https:\/\/sail.usc.edu:\/ccmi\/wp-json\/wp\/v2\/types\/page"}],"author":[{"embeddable":true,"href":"https:\/\/sail.usc.edu:\/ccmi\/wp-json\/wp\/v2\/users\/1"}],"replies":[{"embeddable":true,"href":"https:\/\/sail.usc.edu:\/ccmi\/wp-json\/wp\/v2\/comments?post=54"}],"version-history":[{"count":8,"href":"https:\/\/sail.usc.edu:\/ccmi\/wp-json\/wp\/v2\/pages\/54\/revisions"}],"predecessor-version":[{"id":1317,"href":"https:\/\/sail.usc.edu:\/ccmi\/wp-json\/wp\/v2\/pages\/54\/revisions\/1317"}],"wp:attachment":[{"href":"https:\/\/sail.usc.edu:\/ccmi\/wp-json\/wp\/v2\/media?parent=54"}],"curies":[{"name":"wp","href":"https:\/\/api.w.org\/{rel}","templated":true}]}}