{
  "altLabels": [
    "Massive Text Embedding Benchmark"
  ],
  "artifactKind": "Benchmark",
  "definition": "The Massive Text Embedding Benchmark, a benchmark suite for evaluating text embedding models across multiple tasks, datasets, and languages.",
  "derivedRelations": [
    {
      "note": "BrowseComp-Plus evaluates deep-research agents and retrievers; MTEB evaluates text embedding models across multiple tasks.",
      "target": "browsecomp-plus",
      "type": "notEquivalentTo"
    },
    {
      "note": "MTEB evaluates text embedding models across tasks; web browsing agent evaluation evaluates agent behavior in web information-seeking tasks.",
      "target": "web-browsing-agent-evaluation",
      "type": "notEquivalentTo"
    }
  ],
  "distinctions": [
    {
      "reason": "MTEB evaluates text embeddings across many NLP and retrieval-adjacent tasks; BEIR is an information-retrieval benchmark suite.",
      "statement": "MTEB is not BEIR.",
      "target": "beir",
      "targetLabel": "BEIR",
      "targetUri": "https://id.searchplex.net/beir/"
    },
    {
      "reason": "MTEB is a benchmark suite for evaluating embedding models; an embedding is a vector representation.",
      "statement": "MTEB is not Embedding.",
      "target": "embedding",
      "targetLabel": "Embedding",
      "targetUri": "https://id.searchplex.net/embedding/"
    },
    {
      "reason": "MTEB is a benchmark suite aggregating many tasks and datasets; a test collection is an evaluation resource pattern.",
      "statement": "MTEB is not Test Collection.",
      "target": "test-collection",
      "targetLabel": "Test Collection",
      "targetUri": "https://id.searchplex.net/test-collection/"
    }
  ],
  "id": "mteb",
  "kind": "Artifact",
  "license": "https://creativecommons.org/publicdomain/zero/1.0/",
  "links": [
    {
      "label": "MTEB Leaderboard",
      "type": "officialHomepage",
      "url": "https://huggingface.co/spaces/mteb/leaderboard"
    },
    {
      "label": "MTEB repository",
      "type": "officialRepository",
      "url": "https://github.com/embeddings-benchmark/mteb"
    },
    {
      "label": "MTEB: Massive Text Embedding Benchmark",
      "type": "definingReference",
      "url": "https://arxiv.org/abs/2210.07316"
    }
  ],
  "modifiedAt": "2026-09-14T06:43:46Z",
  "prefLabel": "MTEB",
  "publisher": {
    "name": "Searchplex",
    "url": "https://searchplex.net/"
  },
  "relations": [
    {
      "target": "dense-retrieval",
      "type": "evaluates"
    },
    {
      "target": "embedding",
      "type": "evaluates"
    },
    {
      "target": "reranking",
      "type": "evaluates"
    },
    {
      "target": "retrieval",
      "type": "evaluates"
    },
    {
      "note": "MTEB evaluates text embeddings across many NLP and retrieval-adjacent tasks; BEIR is an information-retrieval benchmark suite.",
      "target": "beir",
      "type": "notEquivalentTo"
    },
    {
      "note": "MTEB is a benchmark suite for evaluating embedding models; an embedding is a vector representation.",
      "target": "embedding",
      "type": "notEquivalentTo"
    },
    {
      "note": "MTEB is a benchmark suite aggregating many tasks and datasets; a test collection is an evaluation resource pattern.",
      "target": "test-collection",
      "type": "notEquivalentTo"
    },
    {
      "target": "beir",
      "type": "related"
    },
    {
      "target": "dense-retrieval",
      "type": "related"
    },
    {
      "target": "embedding",
      "type": "related"
    },
    {
      "target": "reranking",
      "type": "related"
    },
    {
      "target": "retrieval",
      "type": "related"
    },
    {
      "target": "test-collection",
      "type": "related"
    }
  ],
  "representations": {
    "html": "https://id.searchplex.net/mteb/",
    "json": "https://id.searchplex.net/mteb.json",
    "jsonld": "https://id.searchplex.net/mteb.jsonld"
  },
  "scheme": "https://id.searchplex.net/scheme/",
  "scopeNote": "MTEB includes retrieval and reranking tasks, but it is broader than IR evaluation alone because it also covers tasks such as classification, clustering, semantic textual similarity, summarization, pair classification, and bitext mining. It should not be treated as the same kind of benchmark suite as BEIR, which is focused on information retrieval.",
  "status": "published",
  "terms": [
    {
      "community": "ml-nlp",
      "label": "MTEB Leaderboard",
      "usage": "nearSynonym"
    },
    {
      "community": "industry",
      "label": "Embedding Benchmark",
      "usage": "nearSynonym"
    }
  ],
  "uri": "https://id.searchplex.net/mteb/"
}
