edge_id stringlengths 20 78 | source_sense_id stringlengths 9 30 | source_lexeme_id stringlengths 2 15 | headword stringlengths 2 15 | pos stringclasses 9
values | type stringclasses 14
values | target_term stringlengths 1 48 | target_lexeme_id stringlengths 1 48 | target_sense_id stringlengths 9 30 ⌀ | resolved bool 2
classes | confidence float64 0 1 ⌀ | note stringlengths 23 149 ⌀ | tier stringclasses 3
values |
|---|---|---|---|---|---|---|---|---|---|---|---|---|
aaa:noun:0-synonym->nonprofit_organization | aaa:noun:0 | aaa | aaa | noun | synonym | nonprofit organization | nonprofit_organization | null | false | null | null | core |
aaa:noun:0-antonym->for_profit_company | aaa:noun:0 | aaa | aaa | noun | antonym | for profit company | for_profit_company | null | false | null | null | core |
aaa:noun:0-antonym->profit_oriented_organization | aaa:noun:0 | aaa | aaa | noun | antonym | profit oriented organization | profit_oriented_organization | null | false | null | null | core |
aaa:noun:0-hypernym->organization | aaa:noun:0 | aaa | aaa | noun | hypernym | organization | organization | organization:noun:1 | true | 0.95 | null | core |
aaa:noun:1-synonym->top_rating | aaa:noun:1 | aaa | aaa | noun | synonym | top rating | top_rating | null | false | null | null | core |
aaa:noun:1-synonym->prime_rating | aaa:noun:1 | aaa | aaa | noun | synonym | prime rating | prime_rating | null | false | null | null | core |
aaa:noun:1-synonym->highest_credit_quality | aaa:noun:1 | aaa | aaa | noun | synonym | highest credit quality | highest_credit_quality | null | false | null | null | core |
aaa:noun:1-synonym->best_grade | aaa:noun:1 | aaa | aaa | noun | synonym | best grade | best_grade | null | false | null | null | core |
aaa:noun:1-antonym->junk_rating | aaa:noun:1 | aaa | aaa | noun | antonym | junk rating | junk_rating | null | false | null | null | core |
aaa:noun:1-antonym->speculative_grade | aaa:noun:1 | aaa | aaa | noun | antonym | speculative grade | speculative_grade | null | false | null | null | core |
aaa:noun:1-antonym->low_credit_quality | aaa:noun:1 | aaa | aaa | noun | antonym | low credit quality | low_credit_quality | null | false | null | null | core |
aaa:noun:1-hypernym->credit_rating | aaa:noun:1 | aaa | aaa | noun | hypernym | credit rating | credit_rating | null | false | null | null | core |
aaa:noun:2-synonym->triple_a_cell | aaa:noun:2 | aaa | aaa | noun | synonym | triple A cell | triple_a_cell | null | false | null | null | core |
aaa:noun:2-hypernym->small_battery | aaa:noun:2 | aaa | aaa | noun | hypernym | small battery | small_battery | null | false | null | retyped: nano synonym→hypernym | core |
aaa:noun:2-hypernym->battery | aaa:noun:2 | aaa | aaa | noun | hypernym | battery | battery | battery:noun:0 | true | 0.95 | null | core |
aaa:noun:3-antonym->no_authentication | aaa:noun:3 | aaa | aaa | noun | antonym | no authentication | no_authentication | null | false | null | null | core |
aaa:noun:3-hypernym->security_concept | aaa:noun:3 | aaa | aaa | noun | hypernym | security concept | security_concept | null | false | null | null | core |
aaa:interjection:0-synonym->ouch | aaa:interjection:0 | aaa | aaa | interjection | synonym | ouch | ouch | ouch:interjection:0 | true | 0.96 | null | core |
aaa:interjection:0-synonym->ow | aaa:interjection:0 | aaa | aaa | interjection | synonym | ow | ow | null | false | null | null | core |
aaa:interjection:0-hypernym->exclamation | aaa:interjection:0 | aaa | aaa | interjection | hypernym | exclamation | exclamation | exclamation:noun:0 | true | 0.7 | null | core |
aaa:interjection:0-hypernym->expression | aaa:interjection:0 | aaa | aaa | interjection | hypernym | expression | expression | expression:noun:0 | true | 0.95 | null | core |
aaa:interjection:0-hyponym->yelp | aaa:interjection:0 | aaa | aaa | interjection | hyponym | yelp | yelp | yelp:interjection:0 | true | 0.95 | null | core |
aaa:interjection:0-hyponym->groan | aaa:interjection:0 | aaa | aaa | interjection | hyponym | groan | groan | groan:noun:0 | true | 0.55 | null | core |
aaa:interjection:1-synonym->wow | aaa:interjection:1 | aaa | aaa | interjection | synonym | wow | wow | wow:interjection:0 | true | 0.85 | null | core |
aaa:interjection:1-synonym->gosh | aaa:interjection:1 | aaa | aaa | interjection | synonym | gosh | gosh | gosh:interjection:0 | true | 0.85 | null | core |
aaa:interjection:1-hypernym->exclamation | aaa:interjection:1 | aaa | aaa | interjection | hypernym | exclamation | exclamation | exclamation:noun:0 | true | 0.7 | null | core |
aaa:interjection:1-hypernym->expression | aaa:interjection:1 | aaa | aaa | interjection | hypernym | expression | expression | expression:noun:0 | true | 0.95 | null | core |
aaa:interjection:1-hyponym->oh | aaa:interjection:1 | aaa | aaa | interjection | hyponym | oh | oh | null | false | null | null | core |
aaa:interjection:1-hyponym->gasp | aaa:interjection:1 | aaa | aaa | interjection | hyponym | gasp | gasp | gasp:noun:0 | true | 0.95 | null | core |
aaa:interjection:1-hyponym->yikes | aaa:interjection:1 | aaa | aaa | interjection | hyponym | yikes | yikes | yikes:interjection:0 | true | 0.9 | null | core |
aachen:noun:0-synonym->aix_la_chapelle | aachen:noun:0 | aachen | aachen | noun | synonym | Aix-la-Chapelle | aix_la_chapelle | null | false | null | null | tier2 |
aachen:noun:0-hypernym->city | aachen:noun:0 | aachen | aachen | noun | hypernym | city | city | city:noun:2 | true | 0.9 | null | tier2 |
aachen:noun:0-hypernym->municipality | aachen:noun:0 | aachen | aachen | noun | hypernym | municipality | municipality | municipality:noun:0 | true | 0.95 | null | tier2 |
aachen:noun:0-hyponym->university_city | aachen:noun:0 | aachen | aachen | noun | hyponym | university city | university_city | null | false | null | null | tier2 |
aachen:noun:0-hyponym->spa_town | aachen:noun:0 | aachen | aachen | noun | hyponym | spa town | spa_town | null | false | null | null | tier2 |
aachen:noun:0-hypernym->historic_site | aachen:noun:0 | aachen | aachen | noun | hypernym | historic site | historic_site | null | false | null | null | tier2 |
aachen:noun:0-hyponym->carolingian_capital | aachen:noun:0 | aachen | aachen | noun | hyponym | Carolingian capital | carolingian_capital | null | false | null | null | tier2 |
aachen:noun:0-hyponym->coronation_site | aachen:noun:0 | aachen | aachen | noun | hyponym | coronation site | coronation_site | null | false | null | null | tier2 |
aaliyah:noun:0-synonym->forename | aaliyah:noun:0 | aaliyah | aaliyah | noun | synonym | forename | forename | null | false | null | null | tier3 |
aaliyah:noun:0-synonym->personal_name | aaliyah:noun:0 | aaliyah | aaliyah | noun | synonym | personal name | personal_name | null | false | null | null | tier3 |
aaliyah:noun:0-hypernym->name | aaliyah:noun:0 | aaliyah | aaliyah | noun | hypernym | name | name | name:noun:0 | true | 0.95 | null | tier3 |
aaliyah:noun:0-hyponym->arabic_name | aaliyah:noun:0 | aaliyah | aaliyah | noun | hyponym | arabic name | arabic_name | null | false | null | null | tier3 |
aaliyah:noun:0-hyponym->feminine_name | aaliyah:noun:0 | aaliyah | aaliyah | noun | hyponym | feminine name | feminine_name | null | false | null | null | tier3 |
aaliyah:noun:1-synonym->singer | aaliyah:noun:1 | aaliyah | aaliyah | noun | synonym | singer | singer | singer:noun:0 | true | 0.95 | null | tier3 |
aaliyah:noun:1-synonym->musician | aaliyah:noun:1 | aaliyah | aaliyah | noun | synonym | musician | musician | musician:noun:0 | true | 0.9 | null | tier3 |
aaliyah:noun:1-synonym->performer | aaliyah:noun:1 | aaliyah | aaliyah | noun | synonym | performer | performer | performer:noun:0 | true | 0.85 | null | tier3 |
aaliyah:noun:1-hypernym->public_figure | aaliyah:noun:1 | aaliyah | aaliyah | noun | hypernym | public figure | public_figure | null | false | null | null | tier3 |
aaliyah:noun:1-instance_of->artist | aaliyah:noun:1 | aaliyah | aaliyah | noun | instance_of | artist | artist | artist:noun:1 | true | 0.9 | retyped: proper noun instance | tier3 |
aaliyah:noun:1-hypernym->person | aaliyah:noun:1 | aaliyah | aaliyah | noun | hypernym | person | person | person:noun:0 | true | 0.9 | null | tier3 |
aaliyah:noun:1-hyponym->singer | aaliyah:noun:1 | aaliyah | aaliyah | noun | hyponym | singer | singer | singer:noun:0 | true | 0.95 | null | tier3 |
aaliyah:noun:1-hyponym->actress | aaliyah:noun:1 | aaliyah | aaliyah | noun | hyponym | actress | actress | actress:noun:0 | true | 0.95 | null | tier3 |
aaliyah:noun:1-hyponym->recording_artist | aaliyah:noun:1 | aaliyah | aaliyah | noun | hyponym | recording artist | recording_artist | null | false | null | null | tier3 |
aardvark:noun:0-synonym->antbear | aardvark:noun:0 | aardvark | aardvark | noun | synonym | antbear | antbear | null | false | null | null | tier2 |
aardvark:noun:0-synonym->orycteropus | aardvark:noun:0 | aardvark | aardvark | noun | synonym | orycteropus | orycteropus | null | false | null | null | tier2 |
aardvark:noun:0-hyponym->order_tubulidentata | aardvark:noun:0 | aardvark | aardvark | noun | hyponym | order Tubulidentata | order_tubulidentata | null | false | null | retyped: nano hypernym→hyponym | tier2 |
aardvark:noun:0-hypernym->mammal | aardvark:noun:0 | aardvark | aardvark | noun | hypernym | mammal | mammal | mammal:noun:0 | true | 0.95 | null | tier2 |
aardvark:noun:0-hypernym->vertebrate | aardvark:noun:0 | aardvark | aardvark | noun | hypernym | vertebrate | vertebrate | vertebrate:noun:0 | true | 0.95 | null | tier2 |
aardvark:noun:0-hypernym->animal | aardvark:noun:0 | aardvark | aardvark | noun | hypernym | animal | animal | animal:noun:0 | true | 0.93 | null | tier2 |
aardvark:noun:0-synonym->orycteropus_afer | aardvark:noun:0 | aardvark | aardvark | noun | synonym | Orycteropus afer | orycteropus_afer | null | false | null | retyped: nano hyponym→synonym | tier2 |
aaron:noun:0-hypernym->forename | aaron:noun:0 | aaron | aaron | noun | hypernym | forename | forename | null | false | null | retyped: nano synonym→hypernym | core |
aaron:noun:0-hypernym->name | aaron:noun:0 | aaron | aaron | noun | hypernym | name | name | name:noun:0 | true | 0.9 | null | core |
aaron:noun:0-hypernym->personal_name | aaron:noun:0 | aaron | aaron | noun | hypernym | personal name | personal_name | null | false | null | null | core |
aaron:noun:0-synonym->aron | aaron:noun:0 | aaron | aaron | noun | synonym | aron | aron | aron:noun:1 | true | 0.9 | reciprocal of aron:noun:1 | core |
aaron:noun:1-hyponym->high_priest | aaron:noun:1 | aaron | aaron | noun | hyponym | high priest | high_priest | null | false | null | retyped: nano synonym→hyponym | core |
aaron:noun:1-hypernym->historical_figure | aaron:noun:1 | aaron | aaron | noun | hypernym | historical figure | historical_figure | null | false | null | null | core |
aaron:noun:1-hypernym->person | aaron:noun:1 | aaron | aaron | noun | hypernym | person | person | person:noun:0 | true | 0.9 | null | core |
aba:noun:0-synonym->cloak | aba:noun:0 | aba | aba | noun | synonym | cloak | cloak | cloak:noun:0 | true | 0.9 | null | tier2 |
aba:noun:0-synonym->mantle | aba:noun:0 | aba | aba | noun | synonym | mantle | mantle | mantle:noun:0 | true | 0.85 | null | tier2 |
aba:noun:0-hypernym->garment | aba:noun:0 | aba | aba | noun | hypernym | garment | garment | garment:noun:0 | true | 0.9 | null | tier2 |
aba:noun:0-hypernym->outerwear | aba:noun:0 | aba | aba | noun | hypernym | outerwear | outerwear | null | false | null | null | tier2 |
aba:noun:0-synonym->overgarment | aba:noun:0 | aba | aba | noun | synonym | overgarment | overgarment | null | false | null | null | tier2 |
aba:noun:0-synonym->wrap | aba:noun:0 | aba | aba | noun | synonym | wrap | wrap | wrap:noun:1 | true | 0.75 | null | tier2 |
aba:noun:0-hypernym->traditional_garment | aba:noun:0 | aba | aba | noun | hypernym | traditional garment | traditional_garment | null | false | null | null | tier2 |
aba:noun:2-synonym->applied_behavior_analysis | aba:noun:2 | aba | aba | noun | synonym | applied behavior analysis | applied_behavior_analysis | null | false | null | null | tier2 |
aba:noun:2-hypernym->behavior_therapy | aba:noun:2 | aba | aba | noun | hypernym | behavior therapy | behavior_therapy | null | false | null | retyped: nano synonym→hypernym | tier2 |
aba:noun:2-hypernym->behavioral_science | aba:noun:2 | aba | aba | noun | hypernym | behavioral science | behavioral_science | null | false | null | null | tier2 |
aba:noun:2-hypernym->intervention_method | aba:noun:2 | aba | aba | noun | hypernym | intervention method | intervention_method | null | false | null | null | tier2 |
aback:adverb:0-synonym->rearward | aback:adverb:0 | aback | aback | adverb | synonym | rearward | rearward | rearward:adverb:0 | true | 0.9 | null | tier2 |
aback:adverb:0-hypernym->spatial_adverb | aback:adverb:0 | aback | aback | adverb | hypernym | spatial adverb | spatial_adverb | null | false | null | null | tier2 |
aback:adverb:0-synonym->backwind | aback:adverb:0 | aback | aback | adverb | synonym | backwind | backwind | null | false | null | null | tier2 |
aback:adverb:1-synonym->surprised | aback:adverb:1 | aback | aback | adverb | synonym | surprised | surprised | surprised:adjective:0 | true | 0.95 | null | tier2 |
aback:adverb:1-synonym->astonished | aback:adverb:1 | aback | aback | adverb | synonym | astonished | astonished | null | false | null | null | tier2 |
aback:adverb:1-hypernym->emotional_state | aback:adverb:1 | aback | aback | adverb | hypernym | emotional state | emotional_state | null | false | null | null | tier2 |
aback:adverb:1-hypernym->reaction | aback:adverb:1 | aback | aback | adverb | hypernym | reaction | reaction | reaction:noun:0 | true | 0.7 | null | tier2 |
aback:adverb:1-hyponym->astonishment | aback:adverb:1 | aback | aback | adverb | hyponym | astonishment | astonishment | astonishment:noun:0 | true | 0.95 | null | tier2 |
aback:adverb:1-hyponym->surprise | aback:adverb:1 | aback | aback | adverb | hyponym | surprise | surprise | surprise:noun:1 | true | 0.95 | null | tier2 |
aback:adverb:1-hyponym->shock | aback:adverb:1 | aback | aback | adverb | hyponym | shock | shock | shock:noun:0 | true | 0.8 | null | tier2 |
aback:adjective:0-synonym->astonished | aback:adjective:0 | aback | aback | adjective | synonym | astonished | astonished | null | false | null | null | tier2 |
aback:adjective:0-synonym->surprised | aback:adjective:0 | aback | aback | adjective | synonym | surprised | surprised | surprised:adjective:0 | true | 0.9 | null | tier2 |
aback:adjective:0-antonym->unmoved | aback:adjective:0 | aback | aback | adjective | antonym | unmoved | unmoved | unmoved:adjective:1 | true | 0.85 | null | tier2 |
aback:adjective:0-hypernym->emotional_state | aback:adjective:0 | aback | aback | adjective | hypernym | emotional state | emotional_state | null | false | null | null | tier2 |
aback:adjective:0-hyponym->astonished | aback:adjective:0 | aback | aback | adjective | hyponym | astonished | astonished | null | false | null | null | tier2 |
aback:adjective:0-hyponym->startled | aback:adjective:0 | aback | aback | adjective | hyponym | startled | startled | startled:adjective:0 | true | 0.9 | null | tier2 |
aback:adjective:0-hyponym->surprised | aback:adjective:0 | aback | aback | adjective | hyponym | surprised | surprised | surprised:adjective:0 | true | 0.85 | null | tier2 |
aback:adjective:1-synonym->rearward | aback:adjective:1 | aback | aback | adjective | synonym | rearward | rearward | rearward:adjective:1 | true | 0.9 | null | tier2 |
aback:adjective:1-synonym->backward | aback:adjective:1 | aback | aback | adjective | synonym | backward | backward | backward:adjective:0 | true | 0.85 | null | tier2 |
aback:adjective:1-hypernym->orientation | aback:adjective:1 | aback | aback | adjective | hypernym | orientation | orientation | orientation:noun:0 | true | 0.75 | null | tier2 |
aback:adjective:1-hypernym->direction | aback:adjective:1 | aback | aback | adjective | hypernym | direction | direction | direction:noun:0 | true | 0.7 | null | tier2 |
abacus:noun:0-synonym->counting_frame | abacus:noun:0 | abacus | abacus | noun | synonym | counting frame | counting_frame | null | false | null | null | tier2 |
abacus:noun:0-hypernym->calculating_device | abacus:noun:0 | abacus | abacus | noun | hypernym | calculating device | calculating_device | null | false | null | null | tier2 |
Superseded by OpenGloss v2.1 (2026-09-07): 109,633 lexemes and 250,003 live senses — twice this release's coverage — plus a new
opengloss-v2.1-inflectionsform→lemma lookup. v2.0 stays published for reproducibility.
OpenGloss v2.0 — Relations
The OpenGloss v2.0 semantic graph as an edge list. The relations config holds every live typed edge — fourteen relation types — with the target resolved to a sense id wherever the target's entry exists in the release, which is what makes this a sense graph rather than a word graph. The tombstoned config recovers the edges the free reconcile pass demoted, deduplicated or capped away, with the type they carried when they were removed and the reason recorded on them, reconstructed from the provenance trail rather than kept in a side table (D-65, D-68).
Part of the OpenGloss v2.0 release family — 15 datasets built from one store of 54,724 lexemes and 137,314 live senses, all joinable on derived ids. See Related datasets for the rest.
What's new in v2.0 vs v1.3
- Schema v3. Every lexeme carries a
kinddiscriminator (simplex, compound, phrasal verb, idiom, proper noun, abbreviation, affix, function word); every sense carries a controlled domain leaf from a fixed ~160-leaf taxonomy instead of free text; every example carries the character span of the headword occurrence inside it. - Renditions, not one string. A definition is a set: the canonical one plus rewrites at four reading levels and in four registers, each produced in a single call from the canonical text so they say the same thing at different altitudes.
- A sense graph, not a word graph. Typed relations resolve to sense ids wherever
the target's entry exists in the release, so
bank --hypernym--> financial institutionpoints at a meaning rather than at a string. - Retrieval data is first-class. Synthetic per-sense queries in eight styles, grounded QA pairs, mined word-in-context pairs, MS MARCO-style triples with graph-derived hard negatives, and graded TREC qrels — all derivable from, and consistent with, the same entries.
- Derivable identifiers everywhere. v1.3 published a positional id for lexemes and
senses (
3d_model_noun_0) and nothing below that. v2.0 gives every rendition, edge, query, QA pair and provenance record an id computable from the row alone, and never renumbers: a retired sense is tombstoned, so the ids after it keep their meaning. - Per-field provenance. Which model wrote a field, how many tokens it took, what it cost — published as its own dataset.
Scope: fewer headwords, far more per headword
v2.0 is not a superset of v1.3. It covers 54,724 lexemes — a frequency-ranked subset of v1.3's 205,983 — and spends the difference on depth. If you need breadth of vocabulary, use v1.3; if you need graded renditions, resolved relations, spans, or retrieval supervision, use v2.0.
| v1.3 | v2.0 | |
|---|---|---|
| Lexemes | 205,983 | 54,724 |
| Senses | 565,604 | 137,314 |
| Definition renditions per sense | 1 canonical | 1 canonical + up to 8 graded |
| Relation targets | bare strings | resolved to sense ids |
| Retrieval training data | companion sets | queries, QA, triples, qrels |
| Per-field provenance | no | model, tokens and cost per call |
Key statistics
| Lexemes | 54,724 |
| Live senses | 137,314 |
| Rows in this dataset | 1,888,762 |
| Live edges | 735,318 |
| Resolved to a target sense | 390,337 (53.1%) |
| Tombstoned edges recovered | 1,153,444 |
By tier
| Tier | Lexemes | Live senses |
|---|---|---|
core |
10,000 | 34,015 |
tier2 |
31,886 | 76,855 |
tier3 |
12,838 | 26,444 |
Coverage by tier
The release was built in three frequency-ranked passes and they did not all receive the same stages. This table is per-field and per-tier so the gaps are visible rather than averaged away.
| Field | Of | core |
tier2 |
tier3 |
|---|---|---|---|---|
| Canonical gloss | sense | 100.0% | 100.0% | 100.0% |
| Controlled domain tag | sense | 100.0% | 100.0% | 100.0% |
| Gloss at 4 reading levels | sense | 100.0% | 99.9% | 99.9% |
| Gloss in 4 registers | sense | 100.0% | 100.0% | 0.0% |
| At least one example | sense | 100.0% | 99.9% | 99.8% |
| Examples at 4 reading levels | sense | 99.0% | 99.6% | 99.8% |
| At least one relation | sense | 96.8% | 97.3% | 98.0% |
| Synthetic retrieval queries | sense | 100.0% | 100.0% | 0.0% |
| Grounded QA pairs | sense | 99.8% | 99.6% | 0.0% |
| Etymology | lexeme | 100.0% | 100.0% | 99.8% |
| Lexical explanation | lexeme | 100.0% | 100.0% | 100.0% |
| Encyclopedia (neutral) | lexeme | 100.0% | 100.0% | 100.0% |
| Encyclopedia at grade 5 + college (core entries also carry grade 1 and grade 10) | lexeme | 100.0% | 100.0% | 100.0% |
| Contrast paragraphs | lexeme | 72.8% | 55.1% | 0.0% |
Relation types
| Type | Edges |
|---|---|
synonym |
241,860 |
hypernym |
200,988 |
hyponym |
167,405 |
antonym |
103,747 |
meronym |
8,560 |
instance_of |
6,667 |
entails |
2,732 |
causes |
1,108 |
holonym |
1,046 |
derivation |
510 |
used_with |
444 |
collocation |
149 |
see_also |
68 |
confusable_with |
34 |
Tombstoned edges, by reconcile step
| Step | Edges |
|---|---|
tombstone |
929,230 |
cap |
214,570 |
dedup |
9,644 |
Files
| Files | Config | Rows | Shards | Size |
|---|---|---|---|---|
data/relations/train-*.parquet |
relations | 735,318 | 2 | 25.2 MB |
data/tombstoned/train-*.parquet |
tombstoned | 1,153,444 | 3 | 36.4 MB |
Fields
Config relations
735,318 rows, one row per live relation edge.
| Field | Type | Description |
|---|---|---|
edge_id |
string |
Derived edge id: {source_sense_id}-{type}->{target_lexeme_id}. Stable across resolution, because it keys on the target's slug, not its sense. |
source_sense_id |
string |
The sense asserting the relation. |
source_lexeme_id |
string |
That sense's entry id. |
headword |
string |
That entry's headword. |
pos |
string |
Part of speech of the source sense. |
type |
string |
One of synonym, antonym, hypernym, hyponym, meronym, holonym, derivation, collocation, confusable_with, see_also, causes, entails, used_with, instance_of. |
target_term |
string |
The target's surface form, as written. |
target_lexeme_id |
string |
slugify(target_term) — always derivable. |
target_sense_id |
string |
The resolved target sense, or null when the target's entry is not in the release. |
resolved |
bool |
Whether target_sense_id is non-null. |
confidence |
double |
The resolver's confidence, 0–1, when resolved. |
note |
string |
Free text on the edge. Required on confusable_with (how the two differ); elsewhere it carries a hygiene pass's reason. |
tier |
string |
Tier of the source entry. |
One real row:
{
"edge_id": "aaa:noun:0-synonym->nonprofit_organization",
"source_sense_id": "aaa:noun:0",
"source_lexeme_id": "aaa",
"headword": "aaa",
"pos": "noun",
"type": "synonym",
"target_term": "nonprofit organization",
"target_lexeme_id": "nonprofit_organization",
"target_sense_id": null,
"resolved": false,
"confidence": null,
"note": null,
"tier": "core"
}
Config tombstoned
1,153,444 rows, one row per removed relation edge.
| Field | Type | Description |
|---|---|---|
edge_id |
string |
The removed edge's id, reconstructed from the record. |
source_sense_id |
string |
The sense the edge was removed from. |
source_lexeme_id |
string |
That sense's entry id. |
headword |
string |
That entry's headword. |
type |
string |
The type the edge carried when it was removed — which is what its id is built from. The pre-demotion type is only recoverable when the reason names it (retyped: nano synonym→see_also). |
target_term |
string |
The target's surface form. |
target_lexeme_id |
string |
slugify(target_term). |
step |
string |
Which reconcile step removed it: tombstone (a demoted edge), dedup (an exact duplicate) or cap (per-type overflow). |
reason |
string |
The note the edge carried when it was removed — the demotion reason, or - where it had none (usual for cap). |
provenance_id |
string |
The entry provenance record this was read from. |
tier |
string |
Tier of the source entry. |
One real row:
{
"edge_id": "aaa:noun:0-see_also->the_aaa",
"source_sense_id": "aaa:noun:0",
"type": "see_also",
"target_term": "the AAA",
"target_lexeme_id": "the_aaa",
"step": "tombstone",
"reason": "demoted: modifier phrase on headword",
"source_lexeme_id": "aaa",
"headword": "aaa",
"provenance_id": "p58",
"tier": "core"
}
Loading it
from datasets import load_dataset
# configs: "relations", "tombstoned"
ds = load_dataset("mjbommar/opengloss-v2.0-relations", "relations", split="train")
print(ds)
print(ds[0])
The shards are plain parquet, so nothing forces you through datasets — read them
straight, locally or over hf://:
import polars as pl
df = pl.read_parquet("hf://datasets/mjbommar/opengloss-v2.0-relations/data/relations/train-*.parquet")
print(df.head())
import duckdb
duckdb.sql("SELECT count(*) FROM 'hf://datasets/mjbommar/opengloss-v2.0-relations/data/relations/train-*.parquet'").show()
Load the graph into networkx
import networkx as nx
import polars as pl
edges = pl.read_parquet("data/relations/train-*.parquet").filter(
(pl.col("type") == "hypernym") & pl.col("resolved")
)
graph = nx.DiGraph()
graph.add_edges_from(edges.select("source_sense_id", "target_sense_id").rows())
print(graph.number_of_nodes(), "senses,", graph.number_of_edges(), "hypernym edges")
Identifiers, and how they compose
Every id is derived from structure, never randomly minted, so a consumer can recompute one from a row and join across the whole family without a lookup table. Sense positions are stable across regenerations: a retired sense is tombstoned, not removed, so the indices after it never shift.
| Id | Shape | Example |
|---|---|---|
| Lexeme | slugify(headword) |
abseil |
| Sense | {lexeme_id}:{pos}:{index} (zero-based) |
abseil:verb:0 |
| Rendition | {owner_id}#{reading_level}/{register} |
abseil:verb:0#grade_5/plain |
| Entry-level owner | {lexeme_id}:encyclopedia / :explanation |
abseil:encyclopedia |
| Edge | {source_sense_id}-{type}->{target_lexeme_id} |
abseil:verb:0-synonym->rappel |
| Query | {sense_id}#q{n} (zero-based) |
abseil:verb:0#q3 |
| QA pair | {sense_id}#qa{n} (zero-based) |
abseil:verb:0#qa3 |
| Provenance record | p{n} within its entry (one-based) |
p12 |
An edge id keys on the target's slug, not on the target's sense, so resolving a target never changes the id of the edge that found it.
Reading levels and registers
A rendition is keyed on a (reading_level, register) pair. The canonical rendition of
every field is (neutral, plain); everything else is a rewrite of it.
reading_level |
Who it is written for | Rough CCSS band |
|---|---|---|
neutral |
The canonical text: an adult general reader, no level targeted | — |
grade_1 |
Beginning readers; short sentences, common words | K–1 |
grade_5 |
Upper elementary | 4–5 |
grade_10 |
Secondary | 9–10 |
college |
Undergraduate and above; technical vocabulary allowed | 11–CCR |
register |
What changes | Reading it |
|---|---|---|
plain |
Nothing — the neutral register | The default |
informal |
Conversational, contractions, everyday words | How you'd say it to a friend |
formal |
Full forms, precise hedging, no contractions | How you'd write it in a report |
technical |
Domain vocabulary, exact conditions | How a specialist would state it |
marketing |
Benefit-first, persuasive framing | A genre, not a formality level |
marketing sits on the register axis for convenience but is a genre value rather than
a point on the formality scale — worth remembering if you train a formality classifier on
this column.
Related datasets
Everything below is built from the same store and joins on lexeme_id / sense_id.
| Dataset | Grain | What it holds |
|---|---|---|
opengloss-v2.0-lexicon |
one row per lexeme | One row per lexeme: kind, morphology, etymology, encyclopedia, contrasts, sense ids, provenance summary. |
opengloss-v2.0-senses |
one row per live sense | One row per live sense: canonical gloss, 8 gloss renditions, examples, resolved relations, synthetic queries, grounded QA pairs. |
opengloss-v2.0-definitions |
one row per gloss rendition | One row per gloss rendition (canonical included): reading level, register, text, readability grade. |
opengloss-v2.0-examples |
one row per example rendition | One row per example sentence with the headword's character span, its reading level and register. |
opengloss-v2.0-encyclopedia |
one row per encyclopedia rendition · one row per lexical-explanation rendition | One row per encyclopedia article rendition, plus an explanation config for the "why this word" prose. |
opengloss-v2.0-etymology |
one row per entry with an etymology | One row per entry with an etymology: prose summary, ordered language trail, cognates, references. |
opengloss-v2.0-relations (this one) |
one row per live relation edge · one row per removed relation edge | One row per semantic edge, resolved to target sense ids; a tombstoned config recovers the edges the reconcile pass removed. |
opengloss-v2.0-queries |
one row per synthetic query | One row per synthetic retrieval query, across eight query styles, tagged to the sense it should retrieve. |
opengloss-v2.0-qa-pairs |
one row per question/answer pair | One row per grounded question/answer pair, with the rendition ids the answer cites. |
opengloss-v2.0-contrasts |
one row per contrast paragraph | One row per "X vs Y" paragraph on a synonym/antonym/confusable edge, with a verdict on the edge. |
opengloss-v2.0-provenance |
one row per provenance record | One row per recorded generation call: stage, model, tokens, cost, run id — the audit trail. |
opengloss-v2.0-retrieval-pairs |
one row per mined pair | Word-in-context and doc2query-shaped (text_a, text_b, label) pairs mined from the store for free. |
opengloss-v2.0-retrieval-triples |
one row per (query, positive, negative) triple | MS MARCO-style (query, positive, negative) triples whose hard negatives come from the graph. |
opengloss-v2.0-qrels |
one row per query, with its whole graded candidate list · one row per document in the retrieval corpus | Graded TREC relevance judgements (0–3) plus the document corpus and listwise candidate lists. |
opengloss-v2.0-pretrain |
one row per rendered document | Entries serialised into plain-prose dictionary, thesaurus, encyclopedia and usage-note documents. |
Known limitations
- It is synthetic. Every string here was written by a language model against a schema, not transcribed from a corpus or checked by a lexicographer. It is well-formed and internally consistent; it is not attested usage, and it will contain confident errors. Do not use it as ground truth about what a word means.
- Judge scores 70.2/100 (core + tier 2) and 66.7/100 (tier 3). A different model family (Claude Opus) scored fixed 40-entry stratified samples at the close of each build. Sample statistics, not per-entry guarantees, and the judge is itself a model.
- Relation precision is the weakest axis. Relations were judged for validity and the ones that failed were demoted rather than asserted; symmetric reciprocity finished at 98.0% for synonyms and 99.1% for antonyms, and 3,709 senses were left with no relation at all. Treat a single edge as a hypothesis, not a fact; treat the aggregate graph as usable.
- Tier 3 is deliberately partial. 12,838 lexemes received the text stages (glosses, examples, encyclopedia) but not the queries, QA pairs, contrasts or register renditions. The coverage table above gives the exact per-field share; nothing is hidden behind an average.
- The encyclopedia is entry-level. One article per headword, about the headword as a whole. On a polysemous entry it is not a description of any one sense, and it is never used as a positive for one (D-71). It is entry-level reference prose, not a specialist article.
Citation
@misc{bommarito2025opengloss,
title = {OpenGloss: A Synthetic Encyclopedic Dictionary and Semantic Knowledge Graph},
author = {Bommarito, Michael J., II},
year = {2025},
eprint = {2511.18622},
archivePrefix = {arXiv},
url = {https://arxiv.org/abs/2511.18622}
}
License
Released under Creative Commons Attribution 4.0 International (CC-BY 4.0). Attribution to the OpenGloss project is required; commercial use is permitted.
- Downloads last month
- 57