[{"data":1,"prerenderedAt":75},["ShallowReactive",2],{"i-material-symbols:language":3,"i-material-symbols:person":8,"i-mdi:instagram":10,"i-mdi:youtube":12,"i-ri:linkedin-fill":14,"i-ri:twitter-x-line":16,"i-ri:facebook-fill":18,"post-computer-vision":20,"i-material-symbols:person-outline":69,"i-material-symbols:calendar-month":71,"i-mdi:schedule":73},{"left":4,"top":4,"width":5,"height":5,"rotate":4,"vFlip":6,"hFlip":6,"body":7},0,24,false,"\u003Cpath fill=\"currentColor\" d=\"M8.125 21.213q-1.825-.788-3.187-2.15t-2.15-3.188T2 11.988t.788-3.875t2.15-3.175t3.187-2.15T12.013 2t3.875.788t3.175 2.15t2.15 3.175t.787 3.875t-.787 3.887t-2.15 3.188t-3.175 2.15t-3.875.787t-3.888-.787M12 19.95q.65-.9 1.125-1.875T13.9 16h-3.8q.3 1.1.775 2.075T12 19.95m-2.6-.4q-.45-.825-.787-1.713T8.05 16H5.1q.725 1.25 1.813 2.175T9.4 19.55m5.2 0q1.4-.45 2.488-1.375T18.9 16h-2.95q-.225.95-.562 1.838T14.6 19.55M4.25 14h3.4q-.075-.5-.112-.987T7.5 12t.038-1.012T7.65 10h-3.4q-.125.5-.187.988T4 12t.063 1.013t.187.987m5.4 0h4.7q.075-.5.113-.987T14.5 12t-.038-1.012T14.35 10h-4.7q-.075.5-.112.988T9.5 12t.038 1.013t.112.987m6.7 0h3.4q.125-.5.188-.987T20 12t-.062-1.012T19.75 10h-3.4q.075.5.113.988T16.5 12t-.038 1.013t-.112.987m-.4-6h2.95q-.725-1.25-1.812-2.175T14.6 4.45q.45.825.788 1.713T15.95 8M10.1 8h3.8q-.3-1.1-.775-2.075T12 4.05q-.65.9-1.125 1.875T10.1 8m-5 0h2.95q.225-.95.563-1.838T9.4 4.45Q8 4.9 6.912 5.825T5.1 8\"\u002F>",{"left":4,"top":4,"width":5,"height":5,"rotate":4,"vFlip":6,"hFlip":6,"body":9},"\u003Cpath fill=\"currentColor\" d=\"M9.175 10.825Q8 9.65 8 8t1.175-2.825T12 4t2.825 1.175T16 8t-1.175 2.825T12 12t-2.825-1.175M4 20v-2.8q0-.85.438-1.562T5.6 14.55q1.55-.775 3.15-1.162T12 13t3.25.388t3.15 1.162q.725.375 1.163 1.088T20 17.2V20z\"\u002F>",{"left":4,"top":4,"width":5,"height":5,"rotate":4,"vFlip":6,"hFlip":6,"body":11},"\u003Cpath fill=\"currentColor\" d=\"M7.8 2h8.4C19.4 2 22 4.6 22 7.8v8.4a5.8 5.8 0 0 1-5.8 5.8H7.8C4.6 22 2 19.4 2 16.2V7.8A5.8 5.8 0 0 1 7.8 2m-.2 2A3.6 3.6 0 0 0 4 7.6v8.8C4 18.39 5.61 20 7.6 20h8.8a3.6 3.6 0 0 0 3.6-3.6V7.6C20 5.61 18.39 4 16.4 4zm9.65 1.5a1.25 1.25 0 0 1 1.25 1.25A1.25 1.25 0 0 1 17.25 8A1.25 1.25 0 0 1 16 6.75a1.25 1.25 0 0 1 1.25-1.25M12 7a5 5 0 0 1 5 5a5 5 0 0 1-5 5a5 5 0 0 1-5-5a5 5 0 0 1 5-5m0 2a3 3 0 0 0-3 3a3 3 0 0 0 3 3a3 3 0 0 0 3-3a3 3 0 0 0-3-3\"\u002F>",{"left":4,"top":4,"width":5,"height":5,"rotate":4,"vFlip":6,"hFlip":6,"body":13},"\u003Cpath fill=\"currentColor\" d=\"m10 15l5.19-3L10 9zm11.56-7.83c.13.47.22 1.1.28 1.9c.07.8.1 1.49.1 2.09L22 12c0 2.19-.16 3.8-.44 4.83c-.25.9-.83 1.48-1.73 1.73c-.47.13-1.33.22-2.65.28c-1.3.07-2.49.1-3.59.1L12 19c-4.19 0-6.8-.16-7.83-.44c-.9-.25-1.48-.83-1.73-1.73c-.13-.47-.22-1.1-.28-1.9c-.07-.8-.1-1.49-.1-2.09L2 12c0-2.19.16-3.8.44-4.83c.25-.9.83-1.48 1.73-1.73c.47-.13 1.33-.22 2.65-.28c1.3-.07 2.49-.1 3.59-.1L12 5c4.19 0 6.8.16 7.83.44c.9.25 1.48.83 1.73 1.73\"\u002F>",{"left":4,"top":4,"width":5,"height":5,"rotate":4,"vFlip":6,"hFlip":6,"body":15},"\u003Cpath fill=\"currentColor\" d=\"M6.94 5a2 2 0 1 1-4-.002a2 2 0 0 1 4 .002M7 8.48H3V21h4zm6.32 0H9.34V21h3.94v-6.57c0-3.66 4.77-4 4.77 0V21H22v-7.93c0-6.17-7.06-5.94-8.72-2.91z\"\u002F>",{"left":4,"top":4,"width":5,"height":5,"rotate":4,"vFlip":6,"hFlip":6,"body":17},"\u003Cpath fill=\"currentColor\" d=\"M10.488 14.651L15.25 21h7l-7.858-10.478L20.93 3h-2.65l-5.117 5.886L8.75 3h-7l7.51 10.015L2.32 21h2.65zM16.25 19L5.75 5h2l10.5 14z\"\u002F>",{"left":4,"top":4,"width":5,"height":5,"rotate":4,"vFlip":6,"hFlip":6,"body":19},"\u003Cpath fill=\"currentColor\" d=\"M14 13.5h2.5l1-4H14v-2c0-1.03 0-2 2-2h1.5V2.14c-.326-.043-1.557-.14-2.857-.14C11.928 2 10 3.657 10 6.7v2.8H7v4h3V22h4z\"\u002F>",{"postType":21,"id":22,"author":23,"date":26,"modified":27,"title":28,"slug":29,"seoTitle":30,"seoDescription":31,"excerpt":32,"trimmedExcerpt":33,"content":34,"readingTime":35,"featuredImage":36,"postCategory":42,"relatedPosts":47},"post",64661,{"id":24,"name":25},"author-0301142015","Canto","2024-04-16T14:23:21","2026-05-26T17:20:43","Computer vision: How AI models see and understand images","computer-vision","What is computer vision and how does it work?","Find out what computer vision is, exactly how AI models can see and understand images, and where computer vision is headed in the future.","What is computer vision? Computer vision is a field of artificial intelligence that enables computers to interpret and understand the visual world by processing images, video, and other visual inputs. Using deep learning models trained on large labeled image datasets, computer vision systems can identify objects, classify scenes, detect anomalies, and extract structured meaning from [&hellip;]","What is computer vision? Computer vision is a field of artificial intelligence that enables computers to interpret and understand the...","\u003Ch2 class=\"wp-block-heading\" id=\"toc-1--what-is-computer-vision\">What is computer vision?\u003C\u002Fh2>\u003Cp class=\"wp-block-paragraph\">Computer vision is a field of artificial intelligence that enables computers to interpret and understand the visual world by processing images, video, and other visual inputs. Using deep learning models trained on large labeled image datasets, computer vision systems can identify objects, classify scenes, detect anomalies, and extract structured meaning from unstructured visual data. It powers applications ranging from facial recognition and automated quality inspection to AI-driven image search in digital asset libraries.\u003C\u002Fp>\u003Cp class=\"wp-block-paragraph\">\u003Ca href=\"\u002Fglossary\u002Fmarketing-resource-management\u002F\">Marketing resource management\u003C\u002Fa> (MRM) combines tools and processes that enable marketing teams to plan, execute, and measure campaigns with greater efficiency.\u003C\u002Fp>\u003Ch2 class=\"wp-block-heading\" id=\"toc-2--how-does-computer-vision-work\">How does computer vision work?\u003C\u002Fh2>\u003Cp class=\"wp-block-paragraph\">What kind of magic is involved in letting a computer \u003Cem>see\u003C\u002Fem> images? Actually, it is not that magical: to an AI model, a byte of data is a byte of data. It might represent one pixel of an image, or part of a digital sound file, or one character of a text file: it all looks pretty much the same to the computer.\u003C\u002Fp>\u003Cp class=\"wp-block-paragraph\">Certainly, some flavors of AI models work better on images, and others on sound, but for the most part, the ability for an AI model to do a good job at seeing photos and hearing audio is dependent on the training set — a principle that \u003Ca href=\"https:\u002F\u002Fprofiles.stanford.edu\u002Ffei-fei-li\" target=\"_blank\" rel=\"noopener\">Fei-Fei Li\u003C\u002Fa> helped establish through her large-scale labeled image research at Stanford.\u003C\u002Fp>\u003Cp class=\"wp-block-paragraph\">In training, the model looks through your training set over and over. Each time the model looks through your data set is called an \u003Cem>epoch\u003C\u002Fem>, and during each epoch, the model makes small corrections to how each node processes your data until the results stop getting better.\u003C\u002Fp>\u003Cp class=\"wp-block-paragraph\">Then, you graduate to start using the model, feed it an image it has never seen before, and see if you get a correct answer.\u003C\u002Fp>\u003Cp class=\"wp-block-paragraph\">Of course, it is somewhat trickier than this. If you keep training a model with your training set too long, it \u003Cem>overfits\u003C\u002Fem>. Overfitting is when a model does a great job on your training set but when you show it something it hasn’t seen before (but ought to know about), it cannot generalize.\u003C\u002Fp>\u003Cp class=\"wp-block-paragraph\">One of the challenges is understanding why these models work. We have done some work to understand what \u003Cem>excites\u003C\u002Fem> each node inside these models, but we also understand the model, during training, decides for itself what matters in its training set.\u003C\u002Fp>\u003Cp class=\"wp-block-paragraph\">In general, the first layers seem to get excited by large features and then deeper layers pick up finer details, like this:\u003C\u002Fp>\u003Cfigure class=\"wp-block-image aligncenter\">\u003Cimg loading=\"lazy\" decoding=\"async\" width=\"880\" height=\"450\" src=\"\u002Fcdn\u002Fen\u002F2024\u002F04\u002F19183908\u002Fcomputer-vision-layers-nodes-exictations.jpg\" alt=\"An image being analyzed in 4 layers by an AI model over a green background.\" class=\"wp-image-64670\" srcset=\"\u002Fcdn\u002Fen\u002F2024\u002F04\u002F19183908\u002Fcomputer-vision-layers-nodes-exictations.jpg 880w, \u002Fcdn\u002Fen\u002F2024\u002F04\u002F19183908\u002Fcomputer-vision-layers-nodes-exictations-450x230.jpg 450w, \u002Fcdn\u002Fen\u002F2024\u002F04\u002F19183908\u002Fcomputer-vision-layers-nodes-exictations-768x393.jpg 768w\" \u002F>\u003C\u002Ffigure>\u003Cp class=\"wp-block-paragraph\">By 2017, teams competing in the \u003Ca href=\"https:\u002F\u002Fwww.image-net.org\u002F\" target=\"_blank\" rel=\"noopener\">ImageNet\u003C\u002Fa> challenge were hitting 95% accuracy — but in the contest, they only had to make the right choice out of 1,000 categories.\u003C\u002Fp>\u003Cp class=\"wp-block-paragraph\">Let’s move on to how a computer actually understands and categorizes images.\u003C\u002Fp>\u003Ch2 class=\"wp-block-heading\" id=\"toc-3--computer-vision-and-n-dimensional-vectors-explained\">Computer vision and n-dimensional vectors explained\u003C\u002Fh2>\u003Cp class=\"wp-block-paragraph\">How do we get an AI model to \u003Cem>understand\u003C\u002Fem> a complicated picture?\u003C\u002Fp>\u003Cp class=\"wp-block-paragraph\">Complicated images are said to have \u003Cem>high dimensionality\u003C\u002Fem>. In normal life, we are used to three dimensions (like the length, width, and height of your UPS package). However, mathematicians and computers think in ways that are not limited to \u003Cem>normal\u003C\u002Fem>, and so they can mentally work in spaces that have a thousand dimensions or more. Why is that useful?\u003C\u002Fp>\u003Cp class=\"wp-block-paragraph\">Think of a photo of your dog on your couch in your living room, with pictures on the walls, and with a table with magazines on it. This is a fairly normal scene, but can you categorize it into only one category? It is way more complex than that.\u003C\u002Fp>\u003Cp class=\"wp-block-paragraph\">Let’s think about this scene the way that mathematicians and computers do.\u003C\u002Fp>\u003Cp class=\"wp-block-paragraph\">What if you take the photo, and you start with a dot in space? Then, you add an arrow sticking straight up from that dot, and its length is determined by if there is a dog in the photo (there is). Another arrow coming up from that dot but a few degrees to the right indicates if there is a cat in the photo (there is not, so that arrow is length zero). Then an arrow to the left denotes if there is a couch in the photo (there is, so that arrow has some length). An arrow facing down might denote if there is a table in the scene (there is, so that arrow has some length).\u003C\u002Fp>\u003Cp class=\"wp-block-paragraph\">A photo rendered in this way could end up looking like this:\u003C\u002Fp>\u003Cfigure class=\"wp-block-image aligncenter\">\u003Cimg loading=\"lazy\" decoding=\"async\" width=\"880\" height=\"450\" src=\"\u002Fcdn\u002Fen\u002F2024\u002F04\u002F19183907\u002Fcomputer-vision-n-dimensional-vector-digital-fingerprint.jpg\" alt=\"An image of dozens of arrows in different colors representing an n-dimensional vector of a digital image over a blue background.\" class=\"wp-image-64681\" srcset=\"\u002Fcdn\u002Fen\u002F2024\u002F04\u002F19183907\u002Fcomputer-vision-n-dimensional-vector-digital-fingerprint.jpg 880w, \u002Fcdn\u002Fen\u002F2024\u002F04\u002F19183907\u002Fcomputer-vision-n-dimensional-vector-digital-fingerprint-450x230.jpg 450w, \u002Fcdn\u002Fen\u002F2024\u002F04\u002F19183907\u002Fcomputer-vision-n-dimensional-vector-digital-fingerprint-768x393.jpg 768w\" \u002F>\u003C\u002Ffigure>\u003Cp class=\"wp-block-paragraph\">We call this an \u003Cem>n-dimensional vector\u003C\u002Fem>. As you might imagine, if you have enough of these pesky dimensions, you can pretty much represent anything that might be in any photo. Each photo will get its own n-dimensional vector, which you can think of as a \u003Cem>digital fingerprint\u003C\u002Fem>.\u003C\u002Fp>\u003Cp class=\"wp-block-paragraph\">And this digital fingerprint is where things get very useful for anyone searching through photos using AI models trained on vast amounts of images.\u003C\u002Fp>\u003Cp class=\"wp-block-paragraph\">You could find similar looking images by comparing their n-dimensional vectors. If their vectors are similar, the images are likely similar, too.\u003C\u002Fp>\u003Cp class=\"wp-block-paragraph\">That is the next step in computer vision.\u003C\u002Fp>",4,{"url":37,"url_md":38,"alt":39,"height":40,"width":41},"https:\u002F\u002Fwww.canto.com\u002Fcdn\u002Fen\u002F2024\u002F04\u002F19183908\u002Fcomputer-vision-feature.jpg","https:\u002F\u002Fwww.canto.com\u002Fcdn\u002Fen\u002F2024\u002F04\u002F19183908\u002Fcomputer-vision-feature-450x263.jpg","A digital eye iris made of 0s and 1s with images floating around the iris over an orange background.",700,1200,[43],{"id":44,"slug":44,"title":45,"path":46},"ai","Artificial Intelligence","\u002Fpost-category\u002Fai\u002F",[48,55,62],{"featuredImage":49,"postId":50,"slug":51,"title":52,"description":53,"path":54},"https:\u002F\u002Fwww.canto.com\u002Fcdn\u002Fen\u002F2022\u002F09\u002F19184045\u002FCanto_Blog-AI-creativity_feature-450x261.jpg",61999,"how-you-can-make-more-time-for-creativity-the-ai-ready-test","How You Can Make More Time For Creativity: The AI-Ready Test","What would happen if leaders and teams thought about our most-used apps like co-workers? Make more time for creativity using AI tech.","\u002Fblog\u002Fhow-you-can-make-more-time-for-creativity-the-ai-ready-test\u002F",{"featuredImage":56,"postId":57,"slug":58,"title":59,"description":60,"path":61},"https:\u002F\u002Fwww.canto.com\u002Fcdn\u002Fen\u002F2024\u002F06\u002F19183849\u002Fai-nearest-neighbor-feature-450x263.jpg",64825,"descriptors-and-k-nearest-neighbors","How AI discovers content using descriptors and k-nearest neighbors","Discover how the latest AI models use descriptors, also called n-dimensional vectors, and k-nearest neighbors to recommend similar content for today’s consumer.","\u002Fblog\u002Fdescriptors-and-k-nearest-neighbors\u002F",{"featuredImage":63,"postId":64,"slug":65,"title":66,"description":67,"path":68},"https:\u002F\u002Fwww.canto.com\u002Fcdn\u002Fen\u002F2023\u002F04\u002F19184132\u002Fai-content-creation-Feature-450x263.png",63506,"ai-content-creation","How to scale marketing programs using AI content creation","With AI content creation tools, marketers can quickly scale to deliver customized, personalized content to specific audiences to boost engagement.","\u002Fblog\u002Fai-content-creation\u002F",{"left":4,"top":4,"width":5,"height":5,"rotate":4,"vFlip":6,"hFlip":6,"body":70},"\u003Cpath fill=\"currentColor\" d=\"M9.175 10.825Q8 9.65 8 8t1.175-2.825T12 4t2.825 1.175T16 8t-1.175 2.825T12 12t-2.825-1.175M4 20v-2.8q0-.85.438-1.562T5.6 14.55q1.55-.775 3.15-1.162T12 13t3.25.388t3.15 1.162q.725.375 1.163 1.088T20 17.2V20zm2-2h12v-.8q0-.275-.137-.5t-.363-.35q-1.35-.675-2.725-1.012T12 15t-2.775.338T6.5 16.35q-.225.125-.363.35T6 17.2zm7.413-8.587Q14 8.825 14 8t-.587-1.412T12 6t-1.412.588T10 8t.588 1.413T12 10t1.413-.587M12 18\"\u002F>",{"left":4,"top":4,"width":5,"height":5,"rotate":4,"vFlip":6,"hFlip":6,"body":72},"\u003Cpath fill=\"currentColor\" d=\"M12 14q-.425 0-.712-.288T11 13t.288-.712T12 12t.713.288T13 13t-.288.713T12 14m-4.712-.288Q7 13.426 7 13t.288-.712T8 12t.713.288T9 13t-.288.713T8 14t-.712-.288M16 14q-.425 0-.712-.288T15 13t.288-.712T16 12t.713.288T17 13t-.288.713T16 14m-4 4q-.425 0-.712-.288T11 17t.288-.712T12 16t.713.288T13 17t-.288.713T12 18m-4.712-.288Q7 17.426 7 17t.288-.712T8 16t.713.288T9 17t-.288.713T8 18t-.712-.288M16 18q-.425 0-.712-.288T15 17t.288-.712T16 16t.713.288T17 17t-.288.713T16 18M5 22q-.825 0-1.412-.587T3 20V6q0-.825.588-1.412T5 4h1V2h2v2h8V2h2v2h1q.825 0 1.413.588T21 6v14q0 .825-.587 1.413T19 22zm0-2h14V10H5z\"\u002F>",{"left":4,"top":4,"width":5,"height":5,"rotate":4,"vFlip":6,"hFlip":6,"body":74},"\u003Cpath fill=\"currentColor\" d=\"M12 20a8 8 0 0 0 8-8a8 8 0 0 0-8-8a8 8 0 0 0-8 8a8 8 0 0 0 8 8m0-18a10 10 0 0 1 10 10a10 10 0 0 1-10 10C6.47 22 2 17.5 2 12A10 10 0 0 1 12 2m.5 5v5.25l4.5 2.67l-.75 1.23L11 13V7z\"\u002F>",1789448502097]