{
  "altLabels": [
    "Browsing Agent Evaluation",
    "Web Agent Evaluation"
  ],
  "definition": "Evaluation of agents that search, browse, inspect web sources, and synthesize answers or results from information found on the web.",
  "derivedRelations": [
    {
      "target": "browsecomp",
      "type": "evaluatedBy"
    },
    {
      "target": "browsecomp-plus",
      "type": "evaluatedBy"
    }
  ],
  "distinctions": [
    {
      "reason": "BEIR evaluates retrieval systems over fixed IR datasets; web browsing agent evaluation measures agents that search and browse the web or web-like corpora.",
      "statement": "Web Browsing Agent Evaluation is not BEIR.",
      "target": "beir",
      "targetLabel": "BEIR",
      "targetUri": "https://id.searchplex.net/beir/"
    },
    {
      "reason": "MTEB evaluates text embedding models across tasks; web browsing agent evaluation evaluates agent behavior in web information-seeking tasks.",
      "statement": "Web Browsing Agent Evaluation is not MTEB.",
      "target": "mteb",
      "targetLabel": "MTEB",
      "targetUri": "https://id.searchplex.net/mteb/"
    },
    {
      "reason": "Web browsing agent evaluation evaluates agent behavior across search and browsing actions; web search is a search task or system setting.",
      "statement": "Web Browsing Agent Evaluation is not Web Search.",
      "target": "web-search",
      "targetLabel": "Web Search",
      "targetUri": "https://id.searchplex.net/web-search/"
    }
  ],
  "id": "web-browsing-agent-evaluation",
  "kind": "GeneralConcept",
  "license": "https://creativecommons.org/publicdomain/zero/1.0/",
  "links": [
    {
      "label": "BrowseComp",
      "type": "definingReference",
      "url": "https://arxiv.org/abs/2504.12516"
    },
    {
      "label": "BrowseComp-Plus",
      "type": "definingReference",
      "url": "https://arxiv.org/abs/2508.06600"
    }
  ],
  "modifiedAt": "2026-09-14T06:43:46Z",
  "prefLabel": "Web Browsing Agent Evaluation",
  "publisher": {
    "name": "Searchplex",
    "url": "https://searchplex.net/"
  },
  "relations": [
    {
      "note": "BEIR evaluates retrieval systems over fixed IR datasets; web browsing agent evaluation measures agents that search and browse the web or web-like corpora.",
      "target": "beir",
      "type": "notEquivalentTo"
    },
    {
      "note": "MTEB evaluates text embedding models across tasks; web browsing agent evaluation evaluates agent behavior in web information-seeking tasks.",
      "target": "mteb",
      "type": "notEquivalentTo"
    },
    {
      "note": "Web browsing agent evaluation evaluates agent behavior across search and browsing actions; web search is a search task or system setting.",
      "target": "web-search",
      "type": "notEquivalentTo"
    },
    {
      "target": "agentic-retrieval",
      "type": "related"
    },
    {
      "target": "browsecomp",
      "type": "related"
    },
    {
      "target": "browsecomp-plus",
      "type": "related"
    },
    {
      "target": "deep-research",
      "type": "related"
    },
    {
      "target": "web-browsing-agent",
      "type": "related"
    },
    {
      "target": "web-search",
      "type": "related"
    }
  ],
  "representations": {
    "html": "https://id.searchplex.net/web-browsing-agent-evaluation/",
    "json": "https://id.searchplex.net/web-browsing-agent-evaluation.json",
    "jsonld": "https://id.searchplex.net/web-browsing-agent-evaluation.jsonld"
  },
  "scheme": "https://id.searchplex.net/scheme/",
  "scopeNote": "Web browsing agent evaluation may measure source finding, persistence, answer accuracy, evidence use, citation fidelity, and multi-step browsing behavior. It is distinct from ordinary web search evaluation and from retrieval-only benchmark suites.",
  "status": "published",
  "uri": "https://id.searchplex.net/web-browsing-agent-evaluation/"
}
