{"source":"semconv","version":"1.44.0","attribute":{"id":"gen_ai.evaluation.score.label","type":"string","stability":"development","brief":"Human readable label for evaluation.","note":"This attribute provides a human-readable interpretation of the evaluation score produced by an evaluator. For example, a score value of 1 could mean \"relevant\" in one evaluation system and \"not relevant\" in another, depending on the scoring range and evaluator. The label SHOULD have low cardinality. Possible values depend on the evaluation metric and evaluator used; implementations SHOULD document the possible values.","examples":["relevant","not_relevant","correct","incorrect","pass","fail"],"deprecated":{"reason":"uncategorized","note":"Moved to the [OpenTelemetry GenAI semantic conventions repository](https://github.com/open-telemetry/semantic-conventions-genai)."},"namespace":"gen_ai","definedIn":"model/gen-ai/deprecated/registry-deprecated.yaml","usedBy":[{"groupId":"event.gen_ai.evaluation.result","groupType":"event","signalName":"gen_ai.evaluation.result","requirementLevel":"conditionally_required"}]},"history":[{"version":"1.38.0","publishedAt":"2025-10-29T20:23:26Z","kind":"first-seen","detail":"Added as development.","severity":"informational"},{"version":"1.42.0","publishedAt":"2026-06-12T21:03:17Z","kind":"deprecated","detail":"Moved to the [OpenTelemetry GenAI semantic conventions repository](https://github.com/open-telemetry/semantic-conventions-genai).","severity":"notable"}],"replaces":[]}