microsoft
/

OmniParser

+{
+    "bomFormat": "CycloneDX",
+    "specVersion": "1.6",
+    "serialNumber": "urn:uuid:0603eaa6-2ea2-45d9-ba4d-8505097e1c8b",
+    "version": 1,
+    "metadata": {
+        "timestamp": "2025-06-05T09:41:04.764533+00:00",
+        "component": {
+            "type": "machine-learning-model",
+            "bom-ref": "microsoft/OmniParser-c0fec5d3-4871-5122-98a7-dbe046f7ae62",
+            "name": "microsoft/OmniParser",
+            "externalReferences": [
+                {
+                    "url": "https://huggingface.co/microsoft/OmniParser",
+                    "type": "documentation"
+                }
+            ],
+            "modelCard": {
+                "modelParameters": {
+                    "task": "image-text-to-text",
+                    "architectureFamily": "blip-2",
+                    "modelArchitecture": "Blip2ForConditionalGeneration"
+                },
+                "properties": [
+                    {
+                        "name": "library_name",
+                        "value": "transformers"
+                    }
+                ],
+                "consideration": {
+                    "useCases": "- OmniParser is designed to be able to convert unstructured screenshot image into structured list of elements including interactable regions location and captions of icons on its potential functionality.- OmniParser is intended to be used in settings where users are already trained on responsible analytic approaches and critical reasoning is expected. OmniParser is capable of providing extracted information from the screenshot, however human judgement is needed for the output of OmniParser.- OmniParser is intended to be used on various screenshots, which includes both PC and Phone, and also on various applications."
+                }
+            },
+            "authors": [
+                {
+                    "name": "microsoft"
+                }
+            ],
+            "licenses": [
+                {
+                    "license": {
+                        "id": "MIT",
+                        "url": "https://spdx.org/licenses/MIT.html"
+                    }
+                }
+            ],
+            "description": "OmniParser is a general screen parsing tool, which interprets/converts UI screenshot to structured format, to improve existing LLM based UI agent.Training Datasets include: 1) an interactable icon detection dataset, which was curated from popular web pages and automatically annotated to highlight clickable and actionable regions, and 2) an icon description dataset, designed to associate each UI element with its corresponding function.This model hub includes a finetuned version of YOLOv8 and a finetuned BLIP-2 model on the above dataset respectively. For more details of the models used and finetuning, please refer to the [paper](https://arxiv.org/abs/2408.00203).",
+            "tags": [
+                "transformers",
+                "safetensors",
+                "blip-2",
+                "visual-question-answering",
+                "image-text-to-text",
+                "arxiv:2408.00203",
+                "license:mit",
+                "endpoints_compatible",
+                "region:us"
+            ]
+        }
+    }
+}