{"id":{"repo_id":"uoit","oai_identifier":"oai:ontariotechu.scholaris.ca:10155/2049"},"canonical_url":"https://search.dev.ndltd.org/etd/uoit/oai:ontariotechu.scholaris.ca:10155/2049","repository":{"repo_id":"uoit","name":"Ontario Institute of Technology","base_url":"https://ontariotechu.scholaris.ca/server/oai/request"},"display":{"title":"Attention-enhanced Cross-View Transformer for monocular BEV perception","abstract":"Accurately understanding road scenes from monocular images is essential for autonomous driving, yet generating reliable bird&apos;s-eye view (BEV) layouts from single front-view inputs remains challenging. This thesis presents an enhanced cross-view learning framework building upon the PYVA architecture, introducing a Convolutional Block Attention Module (CBAM) to improve feature representation. Our method incorporates spatial-channel attention to refine encoder features, followed by Cycled View Projection (CVP) and a Cross-View Transformer (CVT) for view transformation. CVP enforces geometric coherence through cycle-consistency constraints, while CVT explicitly models feature correspondence across views. We evaluate our approach on KITTI and Argoverse 1.1 datasets for both static and dynamic tasks. Results demonstrate superior performance compared to MonoLayout and PYVA, particularly in mean Average Precision (mAP). The model effectively detects small and distant vehicles while preserving fine road structures in complex scenes.","abstract_html":"Accurately understanding road scenes from monocular images is essential for autonomous driving, yet generating reliable bird&amp;apos;s-eye view (BEV) layouts from single front-view inputs remains challenging. This thesis presents an enhanced cross-view learning framework building upon the PYVA architecture, introducing a Convolutional Block Attention Module (CBAM) to improve feature representation. Our method incorporates spatial-channel attention to refine encoder features, followed by Cycled View Projection (CVP) and a Cross-View Transformer (CVT) for view transformation. CVP enforces geometric coherence through cycle-consistency constraints, while CVT explicitly models feature correspondence across views. We evaluate our approach on KITTI and Argoverse 1.1 datasets for both static and dynamic tasks. Results demonstrate superior performance compared to MonoLayout and PYVA, particularly in mean Average Precision (mAP). The model effectively detects small and distant vehicles while preserving fine road structures in complex scenes.","abstract_has_math":false,"creators":["Jiang, Yu"],"institution":"University of Ontario Institute of Technology","degree_name":"Master of Applied Science (MASc)","degree_level":null,"degree_discipline":"Mechanical Engineering","degree_department":null,"school":null,"contributors":[],"advisors":["Lang, Haoxiang"],"committee_chairs":[],"committee_members":[],"year":2025,"date_issued":"2025-12-01","date_published":"2025-12-01","updated_at":"2026-07-24T05:35:34Z","subjects":[],"languages":["en"],"rights":[],"rights_urls":[],"identifier_entries":[]},"links":{"outbound_url":"https://hdl.handle.net/10155/2049","outbound_label":"Handle","outbound_source":"dc:identifier.uri"},"metadata_groups":[{"id":"people","label":"People","entries":[{"key":"dc:contributor.advisor","label":"Advisor","values":["Lang, Haoxiang"]},{"key":"dc:creator","label":"Author","values":["Jiang, Yu"]}]},{"id":"academic_context","label":"Academic Context","entries":[{"key":"dc:date.accessioned","label":"Dc Date Accessioned","values":["2026-01-20T16:34:12Z"]},{"key":"dc:date.issued","label":"Date","values":["2025-12-01"]},{"key":"dc:type","label":"Dc Type","values":["Thesis"]},{"key":"thesis:degree_discipline","label":"Discipline","values":["Mechanical Engineering"]},{"key":"thesis:degree_name","label":"Degree Name","values":["Master of Applied Science (MASc)"]},{"key":"thesis:institution_name","label":"Thesis Institution Name","values":["University of Ontario Institute of Technology"]}]},{"id":"language_rights","label":"Language and Rights","entries":[{"key":"dc:language.iso","label":"Language (ISO)","values":["en"]}]},{"id":"identifiers","label":"Identifiers","entries":[{"key":"dc:identifier.uri","label":"Identifier URI","values":["https://hdl.handle.net/10155/2049"]}]},{"id":"additional","label":"Additional Metadata","entries":[{"key":"dc:description.abstract","label":"Abstract","values":["Accurately understanding road scenes from monocular images is essential for autonomous driving, yet generating reliable bird&apos;s-eye view (BEV) layouts from single front-view inputs remains challenging. This thesis presents an enhanced cross-view learning framework building upon the PYVA architecture, introducing a Convolutional Block Attention Module (CBAM) to improve feature representation. Our method incorporates spatial-channel attention to refine encoder features, followed by Cycled View Projection (CVP) and a Cross-View Transformer (CVT) for view transformation. CVP enforces geometric coherence through cycle-consistency constraints, while CVT explicitly models feature correspondence across views. We evaluate our approach on KITTI and Argoverse 1.1 datasets for both static and dynamic tasks. Results demonstrate superior performance compared to MonoLayout and PYVA, particularly in mean Average Precision (mAP). The model effectively detects small and distant vehicles while preserving fine road structures in complex scenes."]},{"key":"dc:title","label":"Title","values":["Attention-enhanced Cross-View Transformer for monocular BEV perception"]}]}],"canonical_facts":{"dc:contributor.advisor":["Lang, Haoxiang"],"dc:creator":["Jiang, Yu"],"dc:date.accessioned":["2026-01-20T16:34:12Z"],"dc:date.issued":["2025-12-01"],"dc:description.abstract":["Accurately understanding road scenes from monocular images is essential for autonomous driving, yet generating reliable bird&apos;s-eye view (BEV) layouts from single front-view inputs remains challenging. This thesis presents an enhanced cross-view learning framework building upon the PYVA architecture, introducing a Convolutional Block Attention Module (CBAM) to improve feature representation. Our method incorporates spatial-channel attention to refine encoder features, followed by Cycled View Projection (CVP) and a Cross-View Transformer (CVT) for view transformation. CVP enforces geometric coherence through cycle-consistency constraints, while CVT explicitly models feature correspondence across views. We evaluate our approach on KITTI and Argoverse 1.1 datasets for both static and dynamic tasks. Results demonstrate superior performance compared to MonoLayout and PYVA, particularly in mean Average Precision (mAP). The model effectively detects small and distant vehicles while preserving fine road structures in complex scenes."],"dc:identifier.uri":["https://hdl.handle.net/10155/2049"],"dc:language.iso":["en"],"dc:title":["Attention-enhanced Cross-View Transformer for monocular BEV perception"],"dc:type":["Thesis"],"thesis:degree_discipline":["Mechanical Engineering"],"thesis:degree_name":["Master of Applied Science (MASc)"],"thesis:institution_name":["University of Ontario Institute of Technology"]},"updated_at":"2026-07-24T05:35:34Z"}