@inproceedings{Sabek2026_2468, author = {Sabek, Mohamed and Mei, Qipei and Lee, Gaang and Golabchi, Ali and Gonzalez, Vicente}, editor = {Hamzeh, Farook and Poshdar, Mani and Garcia-Lopez,, Nelly P. and Gan, Vincent}, title = {The Lean Construction Visual Taxonomy (LCVT): bridging the semantic gap}, booktitle = {Proceedings of the 34th Annual Conference of the International Group for Lean Construction (IGLC 34)}, year = {2026}, pages = {14--25}, address = {Singapore, Singapore}, issn = {2789-0015}, doi = {10.24928/2026/0151}, url = {https://www.iglc.net/papers/details/2468}, abstract = {The architecture, engineering, and construction (AEC) industry faces productivity stagnation due to ineffective production flow management. Although Lean Construction (LC) aims to minimize waste, manual monitoring lacks the high-frequency data required for timely control. Computer Vision (CV) offers automated monitoring but suffers from a "Semantic Gap," where models detect low-level objects but fail to interpret high-level Lean states (e.g., "waiting"). This study proposes the Lean Construction Visual Taxonomy (LCVT), a three-level hierarchical framework–Category, Indicator, Visual Definition grounded in Transformation-Flow-Value (TFV) theory. Crucially, the LCVT provides standardized class definitions to guide "zero-shot" prompt engineering in Vision-Language Models (VLMs). By injecting formal L3 definitions that address entity types, temporal thresholds (e.g., stationary >60 s), and spatial context into VLM models such as GPT-4o and Gemini 2.5, the framework enables sophisticated, lean reasoning without the need for massive custom-labeled datasets. Pilot validation achieved a 0.946 mAP in distinguishing state-dependent equipment loads. By formalizing the visual signatures of waste, the LCVT establishes the data infrastructure necessary for proactive, VLM-driven decision support in construction AI.}, keywords = {AI, transformation-flow-value, computer vision, taxonomy, visual management.}, }