{"person":{"slug":"anna-rohrbach","name":"Anna Rohrbach","university":"TU Darmstadt","role":"PI","profile_url":"https://aigude.ai/research/anna-rohrbach"},"generated_at":"2026-09-13T23:11:22.690Z","count":50,"bibtex_url":"https://aigude.ai/api/research/feed/anna-rohrbach?format=bibtex","papers":[{"id":"e9d4e7ab-bbe9-4e99-9d26-0516869dfe62","title":"Think Twice, Act Once: Verifier-Guided Action Selection For Embodied Agents","year":2026,"date":"2026-05-12","venue":"ArXiv.org","venue_slug":null,"venue_type":null,"authors":null,"author_count":7,"doi":null,"arxiv_id":null,"openreview_id":null,"landing_url":"https://arxiv.org/abs/2605.12620","pdf_url":"https://arxiv.org/pdf/2605.12620","links":null,"keywords":null,"tldr":null,"cited_by_count":0,"lab_submitted":false,"register_url":null,"updated_at":"2026-06-24T07:26:35.218175+00:00"},{"id":"e8dd22ab-173c-48a4-985f-2e59edd79d2d","title":"VeriTaS: The First Dynamic Benchmark for Multimodal Automated Fact-Checking","year":2026,"date":null,"venue":"ACL","venue_slug":"acl","venue_type":"main","authors":["Mark Rothermel","Marcus Kornmann","Marcus Rohrbach","Anna Rohrbach"],"author_count":4,"doi":null,"arxiv_id":"2601.08611","openreview_id":null,"landing_url":"https://arxiv.org/abs/2601.08611","pdf_url":"https://arxiv.org/pdf/2601.08611v2","links":null,"keywords":null,"tldr":null,"cited_by_count":null,"lab_submitted":true,"register_url":"https://aigude.ai/research/papers/veritas-the-first-dynamic-benchmark-for-multimodal-automated-fact-checki-2026","updated_at":"2026-09-10T13:51:53.498131+00:00"},{"id":"9e464b7b-3e60-4077-baab-373835608d0c","title":"Chrono: A Simple Blueprint for Representing Time in MLLMs","year":2025,"date":"2025-10-19","venue":"ICCV","venue_slug":"iccv","venue_type":null,"authors":null,"author_count":5,"doi":"10.1109/iccvw69036.2025.00431","arxiv_id":null,"openreview_id":null,"landing_url":"https://doi.org/10.1109/iccvw69036.2025.00431","pdf_url":null,"links":null,"keywords":null,"tldr":null,"cited_by_count":0,"lab_submitted":false,"register_url":null,"updated_at":"2026-07-01T14:17:44.558102+00:00"},{"id":"7d38bf76-1b63-4bd1-843c-78970ad802fa","title":"V² Dial: Unification of Video and Visual Dialog via Multimodal Experts","year":2025,"date":"2025-06-10","venue":"CVPR","venue_slug":"cvpr","venue_type":null,"authors":null,"author_count":4,"doi":"10.1109/cvpr52734.2025.00807","arxiv_id":null,"openreview_id":null,"landing_url":"https://doi.org/10.1109/cvpr52734.2025.00807","pdf_url":null,"links":null,"keywords":null,"tldr":null,"cited_by_count":0,"lab_submitted":false,"register_url":null,"updated_at":"2026-09-10T13:25:11.001811+00:00"},{"id":"e022ad54-d2f9-4d49-8a87-49716647296e","title":"DEFAME: Dynamic Evidence-based FAct-checking with Multimodal Experts.","year":2025,"date":null,"venue":"ICML","venue_slug":"icml","venue_type":null,"authors":null,"author_count":null,"doi":null,"arxiv_id":null,"openreview_id":null,"landing_url":"https://dblp.org/rec/conf/icml/BraunRRR25.html","pdf_url":null,"links":null,"keywords":null,"tldr":null,"cited_by_count":null,"lab_submitted":false,"register_url":null,"updated_at":"2026-07-01T20:27:36.040674+00:00"},{"id":"78ffd55a-53d9-4f95-9bec-b8be9254ab13","title":"Diffusion Classifiers Understand Compositionality, but Conditions Apply.","year":2025,"date":null,"venue":"NeurIPS","venue_slug":"neurips","venue_type":null,"authors":null,"author_count":null,"doi":null,"arxiv_id":null,"openreview_id":null,"landing_url":"https://dblp.org/rec/conf/nips/JeongUOR25.html","pdf_url":null,"links":null,"keywords":null,"tldr":null,"cited_by_count":null,"lab_submitted":false,"register_url":null,"updated_at":"2026-07-01T20:27:36.040674+00:00"},{"id":"d391f398-6c8f-4368-8a67-8a8ee378b0b3","title":"Shape-Guided Diffusion with Inside-Outside Attention.","year":2024,"date":null,"venue":"WACV","venue_slug":"wacv","venue_type":null,"authors":null,"author_count":null,"doi":"10.1109/wacv57701.2024.00415","arxiv_id":null,"openreview_id":null,"landing_url":"https://doi.org/10.1109/wacv57701.2024.00415","pdf_url":null,"links":null,"keywords":null,"tldr":null,"cited_by_count":null,"lab_submitted":false,"register_url":null,"updated_at":"2026-07-01T20:27:36.040674+00:00"},{"id":"b323a58c-29e2-4bd6-87f3-3fb6be7a779e","title":"Simple Token-Level Confidence Improves Caption Correctness.","year":2024,"date":null,"venue":"WACV","venue_slug":"wacv","venue_type":null,"authors":null,"author_count":null,"doi":"10.1109/wacv57701.2024.00564","arxiv_id":null,"openreview_id":null,"landing_url":"https://doi.org/10.1109/wacv57701.2024.00564","pdf_url":null,"links":null,"keywords":null,"tldr":null,"cited_by_count":null,"lab_submitted":false,"register_url":null,"updated_at":"2026-07-01T20:27:36.040674+00:00"},{"id":"d6c3732e-2883-439c-8b94-a4c5e20e2c0a","title":"MammalNet: A Large-Scale Video Benchmark for Mammal Recognition and Behavior Understanding.","year":2023,"date":null,"venue":"CVPR","venue_slug":"cvpr","venue_type":null,"authors":null,"author_count":null,"doi":"10.1109/cvpr52729.2023.01254","arxiv_id":null,"openreview_id":null,"landing_url":"https://doi.org/10.1109/cvpr52729.2023.01254","pdf_url":null,"links":null,"keywords":null,"tldr":null,"cited_by_count":null,"lab_submitted":false,"register_url":null,"updated_at":"2026-07-01T20:27:36.040674+00:00"},{"id":"175d351c-3a30-49de-9ad7-1e1c5d060ec2","title":"More Control for Free! Image Synthesis with Semantic Diffusion Guidance.","year":2023,"date":null,"venue":"WACV","venue_slug":"wacv","venue_type":null,"authors":null,"author_count":null,"doi":"10.1109/wacv56688.2023.00037","arxiv_id":null,"openreview_id":null,"landing_url":"https://doi.org/10.1109/wacv56688.2023.00037","pdf_url":null,"links":null,"keywords":null,"tldr":null,"cited_by_count":null,"lab_submitted":false,"register_url":null,"updated_at":"2026-07-01T20:27:36.040674+00:00"},{"id":"ab69b7ca-9910-4abd-aeb1-5f4411258d7a","title":"Using Language to Extend to Unseen Domains.","year":2023,"date":null,"venue":"ICLR","venue_slug":"iclr","venue_type":null,"authors":null,"author_count":null,"doi":null,"arxiv_id":null,"openreview_id":null,"landing_url":"https://dblp.org/rec/conf/iclr/DunlapMGZDGRR23.html","pdf_url":null,"links":null,"keywords":null,"tldr":null,"cited_by_count":null,"lab_submitted":false,"register_url":null,"updated_at":"2026-07-01T20:27:36.040674+00:00"},{"id":"7ddf87a0-fcaf-4df0-97e8-08963a2eb54d","title":"Watch Those Words: Video Falsification Detection Using Word-Conditioned Facial Motion.","year":2023,"date":null,"venue":"WACV","venue_slug":"wacv","venue_type":null,"authors":null,"author_count":null,"doi":"10.1109/wacv56688.2023.00469","arxiv_id":null,"openreview_id":null,"landing_url":"https://doi.org/10.1109/wacv56688.2023.00469","pdf_url":null,"links":null,"keywords":null,"tldr":null,"cited_by_count":null,"lab_submitted":false,"register_url":null,"updated_at":"2026-07-01T20:27:36.040674+00:00"},{"id":"c86a1da6-eb22-4eca-a1fe-0fc096486793","title":"Bringing Image Scene Structure to Video via Frame-Clip Consistency of Object Tokens.","year":2022,"date":null,"venue":"NeurIPS","venue_slug":"neurips","venue_type":null,"authors":null,"author_count":null,"doi":null,"arxiv_id":null,"openreview_id":null,"landing_url":"https://dblp.org/rec/conf/nips/Ben-AvrahamHMBR22.html","pdf_url":null,"links":null,"keywords":null,"tldr":null,"cited_by_count":null,"lab_submitted":false,"register_url":null,"updated_at":"2026-07-01T20:27:36.040674+00:00"},{"id":"67fd9c98-f2fe-47c5-916b-db8300514a76","title":"DETReg: Unsupervised Pretraining with Region Priors for Object Detection.","year":2022,"date":null,"venue":"CVPR","venue_slug":"cvpr","venue_type":null,"authors":null,"author_count":null,"doi":"10.1109/cvpr52688.2022.01420","arxiv_id":null,"openreview_id":null,"landing_url":"https://doi.org/10.1109/cvpr52688.2022.01420","pdf_url":null,"links":null,"keywords":null,"tldr":null,"cited_by_count":null,"lab_submitted":false,"register_url":null,"updated_at":"2026-07-01T20:27:36.040674+00:00"},{"id":"40e50172-3730-47f0-a880-5bdda5e8f3d9","title":"Exposing the Limits of Video-Text Models through Contrast Sets.","year":2022,"date":null,"venue":"NAACL","venue_slug":"naacl","venue_type":null,"authors":null,"author_count":null,"doi":"10.18653/v1/2022.naacl-main.261","arxiv_id":null,"openreview_id":null,"landing_url":"https://doi.org/10.18653/v1/2022.naacl-main.261","pdf_url":null,"links":null,"keywords":null,"tldr":null,"cited_by_count":null,"lab_submitted":false,"register_url":null,"updated_at":"2026-07-01T20:27:36.040674+00:00"},{"id":"d3fc46ab-8c6e-4843-ac70-958360885e2c","title":"Focus! Relevant and Sufficient Context Selection for News Image Captioning.","year":2022,"date":null,"venue":"EMNLP","venue_slug":"emnlp","venue_type":null,"authors":null,"author_count":null,"doi":"10.18653/v1/2022.findings-emnlp.450","arxiv_id":null,"openreview_id":null,"landing_url":"https://doi.org/10.18653/v1/2022.findings-emnlp.450","pdf_url":null,"links":null,"keywords":null,"tldr":null,"cited_by_count":null,"lab_submitted":false,"register_url":null,"updated_at":"2026-07-01T20:27:36.040674+00:00"},{"id":"83770095-e59c-460f-91b6-7d19fcec2fb4","title":"G3: Geolocation via Guidebook Grounding.","year":2022,"date":null,"venue":"EMNLP","venue_slug":"emnlp","venue_type":null,"authors":null,"author_count":null,"doi":"10.18653/v1/2022.findings-emnlp.430","arxiv_id":null,"openreview_id":null,"landing_url":"https://doi.org/10.18653/v1/2022.findings-emnlp.430","pdf_url":null,"links":null,"keywords":null,"tldr":null,"cited_by_count":null,"lab_submitted":false,"register_url":null,"updated_at":"2026-07-01T20:27:36.040674+00:00"},{"id":"0617cd89-f6a0-40c4-b46c-f52b6797afc8","title":"How Much Can CLIP Benefit Vision-and-Language Tasks?","year":2022,"date":null,"venue":"ICLR","venue_slug":"iclr","venue_type":null,"authors":null,"author_count":null,"doi":null,"arxiv_id":null,"openreview_id":null,"landing_url":"https://dblp.org/rec/conf/iclr/ShenLTBRCYK22.html","pdf_url":null,"links":null,"keywords":null,"tldr":null,"cited_by_count":null,"lab_submitted":false,"register_url":null,"updated_at":"2026-07-01T20:27:36.040674+00:00"},{"id":"5a3bd940-c780-45a5-a9b1-3aa184c89d2e","title":"K-LITE: Learning Transferable Visual Models with External Knowledge.","year":2022,"date":null,"venue":"NeurIPS","venue_slug":"neurips","venue_type":null,"authors":null,"author_count":null,"doi":null,"arxiv_id":null,"openreview_id":null,"landing_url":"https://dblp.org/rec/conf/nips/ShenL0XYZGWY0KD22.html","pdf_url":null,"links":null,"keywords":null,"tldr":null,"cited_by_count":null,"lab_submitted":false,"register_url":null,"updated_at":"2026-07-01T20:27:36.040674+00:00"},{"id":"dd8f80b0-95ab-4474-8cfc-a27aa0f5b822","title":"Object-Region Video Transformers.","year":2022,"date":null,"venue":"CVPR","venue_slug":"cvpr","venue_type":null,"authors":null,"author_count":null,"doi":"10.1109/cvpr52688.2022.00315","arxiv_id":null,"openreview_id":null,"landing_url":"https://doi.org/10.1109/cvpr52688.2022.00315","pdf_url":null,"links":null,"keywords":null,"tldr":null,"cited_by_count":null,"lab_submitted":false,"register_url":null,"updated_at":"2026-07-01T20:27:36.040674+00:00"},{"id":"6dae61c5-f212-4913-9eb9-0893a341e01d","title":"On Guiding Visual Attention with Language Specification.","year":2022,"date":null,"venue":"CVPR","venue_slug":"cvpr","venue_type":null,"authors":null,"author_count":null,"doi":"10.1109/cvpr52688.2022.01756","arxiv_id":null,"openreview_id":null,"landing_url":"https://doi.org/10.1109/cvpr52688.2022.01756","pdf_url":null,"links":null,"keywords":null,"tldr":null,"cited_by_count":null,"lab_submitted":false,"register_url":null,"updated_at":"2026-07-01T20:27:36.040674+00:00"},{"id":"f77bcc1a-d8f3-4e75-a390-f460dccda723","title":"ReCLIP: A Strong Zero-Shot Baseline for Referring Expression Comprehension.","year":2022,"date":null,"venue":"ACL","venue_slug":"acl","venue_type":null,"authors":null,"author_count":null,"doi":"10.18653/v1/2022.acl-long.357","arxiv_id":null,"openreview_id":null,"landing_url":"https://doi.org/10.18653/v1/2022.acl-long.357","pdf_url":null,"links":null,"keywords":null,"tldr":null,"cited_by_count":null,"lab_submitted":false,"register_url":null,"updated_at":"2026-07-01T20:27:36.040674+00:00"},{"id":"86666400-8294-41c7-b153-020361cde46f","title":"Reliable Visual Question Answering: Abstain Rather Than Answer Incorrectly.","year":2022,"date":null,"venue":"ECCV","venue_slug":"eccv","venue_type":null,"authors":null,"author_count":null,"doi":"10.1007/978-3-031-20059-5_9","arxiv_id":null,"openreview_id":null,"landing_url":"https://doi.org/10.1007/978-3-031-20059-5_9","pdf_url":null,"links":null,"keywords":null,"tldr":null,"cited_by_count":null,"lab_submitted":false,"register_url":null,"updated_at":"2026-07-01T20:27:36.040674+00:00"},{"id":"6a7b0d78-0bce-4260-8aeb-64ee0f44b5c8","title":"The Abduction of Sherlock Holmes: A Dataset for Visual Abductive Reasoning.","year":2022,"date":null,"venue":"ECCV","venue_slug":"eccv","venue_type":null,"authors":null,"author_count":null,"doi":"10.1007/978-3-031-20059-5_32","arxiv_id":null,"openreview_id":null,"landing_url":"https://doi.org/10.1007/978-3-031-20059-5_32","pdf_url":null,"links":null,"keywords":null,"tldr":null,"cited_by_count":null,"lab_submitted":false,"register_url":null,"updated_at":"2026-07-01T20:27:36.040674+00:00"},{"id":"2b4452e9-588b-4dc0-aaa2-322ae4a88597","title":"TL;DW? Summarizing Instructional Videos with Task Relevance and Cross-Modal Saliency.","year":2022,"date":null,"venue":"ECCV","venue_slug":"eccv","venue_type":null,"authors":null,"author_count":null,"doi":"10.1007/978-3-031-19830-4_31","arxiv_id":null,"openreview_id":null,"landing_url":"https://doi.org/10.1007/978-3-031-19830-4_31","pdf_url":null,"links":null,"keywords":null,"tldr":null,"cited_by_count":null,"lab_submitted":false,"register_url":null,"updated_at":"2026-07-01T20:27:36.040674+00:00"},{"id":"d85b61bb-8227-44b9-b3c2-41eb9dff915b","title":"Twitter-COMMs: Detecting Climate, COVID, and Military Multimodal Misinformation.","year":2022,"date":null,"venue":"NAACL","venue_slug":"naacl","venue_type":null,"authors":null,"author_count":null,"doi":"10.18653/v1/2022.naacl-main.110","arxiv_id":null,"openreview_id":null,"landing_url":"https://doi.org/10.18653/v1/2022.naacl-main.110","pdf_url":null,"links":null,"keywords":null,"tldr":null,"cited_by_count":null,"lab_submitted":false,"register_url":null,"updated_at":"2026-07-01T20:27:36.040674+00:00"},{"id":"42536540-62fb-4aba-b567-0120ae1519a3","title":"Benchmark for Compositional Text-to-Image Synthesis.","year":2021,"date":null,"venue":"NeurIPS","venue_slug":"neurips","venue_type":null,"authors":null,"author_count":null,"doi":null,"arxiv_id":null,"openreview_id":null,"landing_url":"https://dblp.org/rec/conf/nips/ParkALDR21.html","pdf_url":null,"links":null,"keywords":null,"tldr":null,"cited_by_count":null,"lab_submitted":false,"register_url":null,"updated_at":"2026-07-01T20:27:36.040674+00:00"},{"id":"08794a2a-af08-4649-9c60-ecdbbed49156","title":"CLIP-It! Language-Guided Video Summarization.","year":2021,"date":null,"venue":"NeurIPS","venue_slug":"neurips","venue_type":null,"authors":null,"author_count":null,"doi":null,"arxiv_id":null,"openreview_id":null,"landing_url":"https://dblp.org/rec/conf/nips/NarasimhanRD21.html","pdf_url":null,"links":null,"keywords":null,"tldr":null,"cited_by_count":null,"lab_submitted":false,"register_url":null,"updated_at":"2026-07-01T20:27:36.040674+00:00"},{"id":"e9120b5d-200e-4621-8385-33be6dedf348","title":"Compositional Video Synthesis with Action Graphs.","year":2021,"date":null,"venue":"ICML","venue_slug":"icml","venue_type":null,"authors":null,"author_count":null,"doi":null,"arxiv_id":null,"openreview_id":null,"landing_url":"https://dblp.org/rec/conf/icml/BarH0RCDG21.html","pdf_url":null,"links":null,"keywords":null,"tldr":null,"cited_by_count":null,"lab_submitted":false,"register_url":null,"updated_at":"2026-07-01T20:27:36.040674+00:00"},{"id":"e3de9f2f-7c3f-4771-bb8f-ad4331dccf0e","title":"NewsCLIPpings: Automatic Generation of Out-of-Context Multimodal Media.","year":2021,"date":null,"venue":"EMNLP","venue_slug":"emnlp","venue_type":null,"authors":null,"author_count":null,"doi":"10.18653/v1/2021.emnlp-main.545","arxiv_id":null,"openreview_id":null,"landing_url":"https://doi.org/10.18653/v1/2021.emnlp-main.545","pdf_url":null,"links":null,"keywords":null,"tldr":null,"cited_by_count":null,"lab_submitted":false,"register_url":null,"updated_at":"2026-07-01T20:27:36.040674+00:00"},{"id":"363fbc76-6b28-4998-8df5-90226eba1aaf","title":"Advisable Learning for Self-Driving Vehicles by Internalizing Observation-to-Action Rules.","year":2020,"date":null,"venue":"CVPR","venue_slug":"cvpr","venue_type":null,"authors":null,"author_count":null,"doi":"10.1109/cvpr42600.2020.00968","arxiv_id":null,"openreview_id":null,"landing_url":"https://doi.org/10.1109/cvpr42600.2020.00968","pdf_url":null,"links":null,"keywords":null,"tldr":null,"cited_by_count":null,"lab_submitted":false,"register_url":null,"updated_at":"2026-07-01T20:27:36.040674+00:00"},{"id":"22048cb8-41cd-4c5e-afcc-320326db54e7","title":"Identity-Aware Multi-sentence Video Description.","year":2020,"date":null,"venue":"ECCV","venue_slug":"eccv","venue_type":null,"authors":null,"author_count":null,"doi":"10.1007/978-3-030-58589-1_22","arxiv_id":null,"openreview_id":null,"landing_url":"https://doi.org/10.1007/978-3-030-58589-1_22","pdf_url":null,"links":null,"keywords":null,"tldr":null,"cited_by_count":null,"lab_submitted":false,"register_url":null,"updated_at":"2026-07-01T20:27:36.040674+00:00"},{"id":"a3720787-8cb8-42c9-92c4-0a87eb2f419c","title":"Adversarial Inference for Multi-Sentence Video Description.","year":2019,"date":null,"venue":"CVPR","venue_slug":"cvpr","venue_type":null,"authors":null,"author_count":null,"doi":null,"arxiv_id":null,"openreview_id":null,"landing_url":"https://dblp.org/rec/conf/cvpr/ParkRDR19a.html","pdf_url":null,"links":null,"keywords":null,"tldr":null,"cited_by_count":null,"lab_submitted":false,"register_url":null,"updated_at":"2026-07-01T20:27:36.040674+00:00"},{"id":"8b6e75a6-005d-4dba-bc47-e81d53b8daa6","title":"Are You Looking? Grounding to Multiple Modalities in Vision-and-Language Navigation.","year":2019,"date":null,"venue":"ACL","venue_slug":"acl","venue_type":null,"authors":null,"author_count":null,"doi":"10.18653/v1/p19-1655","arxiv_id":null,"openreview_id":null,"landing_url":"https://doi.org/10.18653/v1/p19-1655","pdf_url":null,"links":null,"keywords":null,"tldr":null,"cited_by_count":null,"lab_submitted":false,"register_url":null,"updated_at":"2026-07-01T20:27:36.040674+00:00"},{"id":"2d627f08-9613-405b-9d14-47b8f9d3dfbe","title":"Language-Conditioned Graph Networks for Relational Reasoning.","year":2019,"date":null,"venue":"ICCV","venue_slug":"iccv","venue_type":null,"authors":null,"author_count":null,"doi":"10.1109/iccv.2019.01039","arxiv_id":null,"openreview_id":null,"landing_url":"https://doi.org/10.1109/iccv.2019.01039","pdf_url":null,"links":null,"keywords":null,"tldr":null,"cited_by_count":null,"lab_submitted":false,"register_url":null,"updated_at":"2026-07-01T20:27:36.040674+00:00"},{"id":"0117dfd5-ee77-4414-92d3-098d86ab6313","title":"Robust Change Captioning.","year":2019,"date":null,"venue":"ICCV","venue_slug":"iccv","venue_type":null,"authors":null,"author_count":null,"doi":"10.1109/iccv.2019.00472","arxiv_id":null,"openreview_id":null,"landing_url":"https://doi.org/10.1109/iccv.2019.00472","pdf_url":null,"links":null,"keywords":null,"tldr":null,"cited_by_count":null,"lab_submitted":false,"register_url":null,"updated_at":"2026-07-01T20:27:36.040674+00:00"},{"id":"c7f078ec-0851-4075-9dfa-7c027ea5ebd9","title":"Fooling Vision and Language Models Despite Localization and Attention Mechanism.","year":2018,"date":null,"venue":"CVPR","venue_slug":"cvpr","venue_type":null,"authors":null,"author_count":null,"doi":"10.1109/cvpr.2018.00520","arxiv_id":null,"openreview_id":null,"landing_url":"https://doi.org/10.1109/cvpr.2018.00520","pdf_url":null,"links":null,"keywords":null,"tldr":null,"cited_by_count":null,"lab_submitted":false,"register_url":null,"updated_at":"2026-07-01T20:27:36.040674+00:00"},{"id":"6a1ba2e7-163c-4eac-a767-5ca2743b1f1a","title":"Multimodal Explanations: Justifying Decisions and Pointing to the Evidence.","year":2018,"date":null,"venue":"CVPR","venue_slug":"cvpr","venue_type":null,"authors":null,"author_count":null,"doi":"10.1109/cvpr.2018.00915","arxiv_id":null,"openreview_id":null,"landing_url":"https://doi.org/10.1109/cvpr.2018.00915","pdf_url":null,"links":null,"keywords":null,"tldr":null,"cited_by_count":null,"lab_submitted":false,"register_url":null,"updated_at":"2026-07-01T20:27:36.040674+00:00"},{"id":"2318876d-2014-4bf2-b257-24c8df822f06","title":"Object Hallucination in Image Captioning.","year":2018,"date":null,"venue":"EMNLP","venue_slug":"emnlp","venue_type":null,"authors":null,"author_count":null,"doi":"10.18653/v1/d18-1437","arxiv_id":null,"openreview_id":null,"landing_url":"https://doi.org/10.18653/v1/d18-1437","pdf_url":null,"links":null,"keywords":null,"tldr":null,"cited_by_count":null,"lab_submitted":false,"register_url":null,"updated_at":"2026-07-01T20:27:36.040674+00:00"},{"id":"be5e3b37-95b5-4b44-b5db-8a72e3c3fc86","title":"Speaker-Follower Models for Vision-and-Language Navigation.","year":2018,"date":null,"venue":"NeurIPS","venue_slug":"neurips","venue_type":null,"authors":null,"author_count":null,"doi":null,"arxiv_id":null,"openreview_id":null,"landing_url":"https://dblp.org/rec/conf/nips/FriedHCRAMBSKD18.html","pdf_url":null,"links":null,"keywords":null,"tldr":null,"cited_by_count":null,"lab_submitted":false,"register_url":null,"updated_at":"2026-07-01T20:27:36.040674+00:00"},{"id":"ac31aad8-385e-42dd-869c-92fae404d16b","title":"Textual Explanations for Self-Driving Vehicles.","year":2018,"date":null,"venue":"ECCV","venue_slug":"eccv","venue_type":null,"authors":null,"author_count":null,"doi":"10.1007/978-3-030-01216-8_35","arxiv_id":null,"openreview_id":null,"landing_url":"https://doi.org/10.1007/978-3-030-01216-8_35","pdf_url":null,"links":null,"keywords":null,"tldr":null,"cited_by_count":null,"lab_submitted":false,"register_url":null,"updated_at":"2026-07-01T20:27:36.040674+00:00"},{"id":"411d5857-cedc-4f1e-a01c-93a6f841c43a","title":"Video Object Segmentation with Referring Expressions.","year":2018,"date":null,"venue":"ECCV","venue_slug":"eccv","venue_type":null,"authors":null,"author_count":null,"doi":"10.1007/978-3-030-11018-5_2","arxiv_id":null,"openreview_id":null,"landing_url":"https://doi.org/10.1007/978-3-030-11018-5_2","pdf_url":null,"links":null,"keywords":null,"tldr":null,"cited_by_count":null,"lab_submitted":false,"register_url":null,"updated_at":"2026-07-01T20:27:36.040674+00:00"},{"id":"4a07e12f-1cb0-44a1-ae62-db016b56436c","title":"Women Also Snowboard: Overcoming Bias in Captioning Models.","year":2018,"date":null,"venue":"ECCV","venue_slug":"eccv","venue_type":null,"authors":null,"author_count":null,"doi":"10.1007/978-3-030-01219-9_47","arxiv_id":null,"openreview_id":null,"landing_url":"https://doi.org/10.1007/978-3-030-01219-9_47","pdf_url":null,"links":null,"keywords":null,"tldr":null,"cited_by_count":null,"lab_submitted":false,"register_url":null,"updated_at":"2026-07-01T20:27:36.040674+00:00"},{"id":"6fd78d18-9335-4901-a614-3abdc4cb9a75","title":"A Dataset and Exploration of Models for Understanding Video Data through Fill-in-the-Blank Question-Answering.","year":2017,"date":null,"venue":"CVPR","venue_slug":"cvpr","venue_type":null,"authors":null,"author_count":null,"doi":"10.1109/cvpr.2017.778","arxiv_id":null,"openreview_id":null,"landing_url":"https://doi.org/10.1109/cvpr.2017.778","pdf_url":null,"links":null,"keywords":null,"tldr":null,"cited_by_count":null,"lab_submitted":false,"register_url":null,"updated_at":"2026-07-01T20:27:36.040674+00:00"},{"id":"0c8578a6-1e49-429c-88bb-8d2eff00e52f","title":"Generating Descriptions with Grounded and Co-referenced People.","year":2017,"date":null,"venue":"CVPR","venue_slug":"cvpr","venue_type":null,"authors":null,"author_count":null,"doi":null,"arxiv_id":null,"openreview_id":null,"landing_url":"https://dblp.org/rec/conf/cvpr/RohrbachRTOS17.html","pdf_url":null,"links":null,"keywords":null,"tldr":null,"cited_by_count":null,"lab_submitted":false,"register_url":null,"updated_at":"2026-07-01T20:27:36.040674+00:00"},{"id":"7ae1848b-6bd5-4da9-9ca3-8f6ccfad21ce","title":"Gradient-free Policy Architecture Search and Adaptation.","year":2017,"date":null,"venue":"CoRL","venue_slug":"corl","venue_type":null,"authors":null,"author_count":null,"doi":null,"arxiv_id":null,"openreview_id":null,"landing_url":"https://dblp.org/rec/conf/corl/EbrahimiRD17.html","pdf_url":null,"links":null,"keywords":null,"tldr":null,"cited_by_count":null,"lab_submitted":false,"register_url":null,"updated_at":"2026-07-01T20:27:36.040674+00:00"},{"id":"cba51c5d-7df5-451c-a30e-09e99249226a","title":"Commonsense in Parts: Mining Part-Whole Relations from the Web and Image Tags.","year":2016,"date":null,"venue":"AAAI","venue_slug":"aaai","venue_type":null,"authors":null,"author_count":null,"doi":"10.1609/aaai.v30i1.9992","arxiv_id":null,"openreview_id":null,"landing_url":"https://doi.org/10.1609/aaai.v30i1.9992","pdf_url":null,"links":null,"keywords":null,"tldr":null,"cited_by_count":null,"lab_submitted":false,"register_url":null,"updated_at":"2026-07-01T20:27:36.040674+00:00"},{"id":"180dc2cf-7de5-41f3-ae10-0a298c83eef8","title":"Grounding of Textual Phrases in Images by Reconstruction.","year":2016,"date":null,"venue":"ECCV","venue_slug":"eccv","venue_type":null,"authors":null,"author_count":null,"doi":"10.1007/978-3-319-46448-0_49","arxiv_id":null,"openreview_id":null,"landing_url":"https://doi.org/10.1007/978-3-319-46448-0_49","pdf_url":null,"links":null,"keywords":null,"tldr":null,"cited_by_count":null,"lab_submitted":false,"register_url":null,"updated_at":"2026-07-01T20:27:36.040674+00:00"},{"id":"2ea646bb-f39a-4c89-90d2-0eae77b74d88","title":"Multimodal Compact Bilinear Pooling for Visual Question Answering and Visual Grounding.","year":2016,"date":null,"venue":"EMNLP","venue_slug":"emnlp","venue_type":null,"authors":null,"author_count":null,"doi":"10.18653/v1/d16-1044","arxiv_id":null,"openreview_id":null,"landing_url":"https://doi.org/10.18653/v1/d16-1044","pdf_url":null,"links":null,"keywords":null,"tldr":null,"cited_by_count":null,"lab_submitted":false,"register_url":null,"updated_at":"2026-07-01T20:27:36.040674+00:00"},{"id":"8953ee05-e98e-4609-8cc2-754f47dd6a83","title":"A dataset for Movie Description.","year":2015,"date":null,"venue":"CVPR","venue_slug":"cvpr","venue_type":null,"authors":null,"author_count":null,"doi":"10.1109/cvpr.2015.7298940","arxiv_id":null,"openreview_id":null,"landing_url":"https://doi.org/10.1109/cvpr.2015.7298940","pdf_url":null,"links":null,"keywords":null,"tldr":null,"cited_by_count":null,"lab_submitted":false,"register_url":null,"updated_at":"2026-07-01T20:27:36.040674+00:00"}]}