{"analysis_contract_sha256":"cf75d7ed5dbb8022756ef7a00652f3f3e991013061ba1d4e2b967139d026ee03","analysis_id":"analysis.delve-index-v1","calculations":[{"calculation_id":"research.count-v1","confidence":"deterministic","description":"Count records, glosses, and exact lowercase a-z token runs.","formula":"count(observations)","limitations":[],"source_fields":["definition_glosses","normalized_spelling","frequency_tier"]},{"calculation_id":"research.share-v1","confidence":"deterministic","description":"Calculate a cohort's exact share of its named denominator.","formula":"100 * subset_count / cohort_count","limitations":["Token shares and record shares use different denominators, both published."],"source_fields":["definition_glosses","normalized_spelling","frequency_tier"]},{"calculation_id":"delve.rate-per-10k-v1","confidence":"deterministic","description":"Normalize exact token occurrences per ten thousand gloss tokens.","formula":"10000 * token_occurrences / total_gloss_tokens","limitations":[],"source_fields":["definition_glosses"]},{"calculation_id":"delve.style-marker-count-v1","confidence":"heuristic","description":"Count exact occurrences and distinct documented headwords for a fixed version-controlled marker list associated with machine-written explanatory prose, publishing zero counts; the combined headword total uses a set union.","formula":"count(token occurrences where token in declared marker list); count(distinct documented headwords containing one or more declared markers)","limitations":["Marker-list membership is a declared editorial heuristic, not evidence about any individual gloss."],"source_fields":["definition_glosses"]},{"calculation_id":"delve.token-leader-v1","confidence":"deterministic","description":"Select the most-frequent rarer-band gloss token, breaking equal counts by ascending token.","formula":"argmax(gloss_occurrences, tie_break=token_ascending)","limitations":[],"source_fields":["definition_glosses","frequency_tier"]}],"charts":[{"accessible_summary":"Exact shares and denominators appear in the adjacent table.","category_column_id":"cohort","chart_id":"study.delve-index.tier-skew-chart","chart_type":"grouped-bar","dataset_id":"study.delve-index.tier-skew","heading":"Gloss token share versus safe record share by tier","series":[{"label":"Gloss tokens (%)","value_column_id":"token_share_percent"},{"label":"Safe records (%)","value_column_id":"record_share_percent"}],"table_id":"study.delve-index.tier-skew-table"}],"citations":[{"citation_id":"citation.first-party-definitions","citation_kind":"dataset","data_version":"2026.07-lexical.6","publisher":"The Word Index","source_id":null,"title":"The Word Index first-party definition corpus","url":"/about/data","version":"2026.07-lexical.6"},{"citation_id":"citation.lexical-sources","citation_kind":"dataset","data_version":"2026.07-lexical.6","publisher":"The Word Index","source_id":null,"title":"The Word Index pinned lexical source manifest","url":"/about/data","version":"2026.07-lexical.6"},{"citation_id":"citation.frequencywords","citation_kind":"dataset","data_version":null,"publisher":"FrequencyWords","source_id":"frequencywords-en-2018","title":"FrequencyWords English 2018","url":"https://raw.githubusercontent.com/hermitdave/FrequencyWords/525f9b560de45753a5ea01069454e72e9aa541c6/content/2018/en/en_50k.txt","version":"git:525f9b560de45753a5ea01069454e72e9aa541c6; corpus-release:2018"},{"citation_id":"citation.content-policy","citation_kind":"policy","data_version":null,"publisher":"The Word Index","source_id":"offensive-term-policy","title":"The Word Index content-safety exclusion policy","url":"/about/data","version":"2026.07-review-pending.2"}],"concise_result":"First-party glosses document 59.98% of safe records; the 22 declared style markers occur 846 times in 2,973,912 gloss tokens.","content_artifact_digest":"477542693badcafd96a49a013917cb986d413ee48d738561b13d04fa8f330416","data_version":"2026.07-lexical.6","datasets":[{"columns":[{"calculation_id":null,"column_id":"row_id","label":"Row ID","role":"dimension","unit":null,"value_type":"text"},{"calculation_id":null,"column_id":"stage","label":"Population stage","role":"dimension","unit":null,"value_type":"text"},{"calculation_id":"research.count-v1","column_id":"record_count","label":"Records","role":"measure","unit":"records","value_type":"integer"}],"dataset_id":"study.delve-index.population","description":"Exact record counts before and after the declared product-policy and study filters.","primary_download":false,"rows":[{"record_count":394595,"row_id":"raw","stage":"Raw lexical ledger"},{"record_count":150,"row_id":"policy-excluded","stage":"Excluded by product policy"},{"record_count":394445,"row_id":"safe","stage":"Safe normalized-spelling records"},{"record_count":394445,"row_id":"analysis","stage":"Records meeting this study's inclusion criteria"}],"title":"Study population"},{"columns":[{"calculation_id":null,"column_id":"row_id","label":"Row ID","role":"dimension","unit":null,"value_type":"text"},{"calculation_id":null,"column_id":"metric","label":"Metric","role":"dimension","unit":null,"value_type":"text"},{"calculation_id":"research.count-v1","column_id":"observation_count","label":"Observations","role":"measure","unit":"observations","value_type":"integer"}],"dataset_id":"study.delve-index.corpus","description":"Exact corpus sizes behind every rate in this self-audit.","primary_download":false,"rows":[{"metric":"Safe records with at least one first-party gloss","observation_count":236598,"row_id":"documented-headwords"},{"metric":"First-party definition glosses","observation_count":271328,"row_id":"definition-glosses"},{"metric":"Gloss token occurrences (a-z runs)","observation_count":2973912,"row_id":"gloss-token-occurrences"},{"metric":"Distinct gloss tokens","observation_count":103776,"row_id":"distinct-gloss-tokens"},{"metric":"Distinct gloss tokens that are safe compiled records","observation_count":99627,"row_id":"tokens-matching-safe-records"}],"title":"Definition-corpus inventory"},{"columns":[{"calculation_id":null,"column_id":"row_id","label":"Row ID","role":"dimension","unit":null,"value_type":"text"},{"calculation_id":null,"column_id":"coverage_class","label":"Coverage class","role":"dimension","unit":null,"value_type":"text"},{"calculation_id":"research.count-v1","column_id":"record_count","label":"Records","role":"measure","unit":"records","value_type":"integer"},{"calculation_id":"research.share-v1","column_id":"share_percent","label":"Share","role":"measure","unit":"percent","value_type":"decimal-string"}],"dataset_id":"study.delve-index.gloss-coverage","description":"Which safe records the first-party definition corpus documents.","primary_download":false,"rows":[{"coverage_class":"Records with first-party glosses","record_count":236598,"row_id":"documented","share_percent":"59.98"},{"coverage_class":"Records without first-party glosses","record_count":157847,"row_id":"undocumented","share_percent":"40.02"}],"title":"Gloss coverage of the safe population"},{"columns":[{"calculation_id":null,"column_id":"row_id","label":"Row ID","role":"dimension","unit":null,"value_type":"text"},{"calculation_id":null,"column_id":"cohort","label":"Cohort","role":"dimension","unit":null,"value_type":"text"},{"calculation_id":"research.count-v1","column_id":"token_occurrences","label":"Gloss token occurrences","role":"measure","unit":"tokens","value_type":"integer"},{"calculation_id":"research.share-v1","column_id":"token_share_percent","label":"Share of gloss tokens","role":"measure","unit":"percent","value_type":"decimal-string"},{"calculation_id":"research.count-v1","column_id":"safe_record_count","label":"Safe records at tier","role":"measure","unit":"records","value_type":"integer"},{"calculation_id":"research.share-v1","column_id":"record_share_percent","label":"Share of safe records","role":"measure","unit":"percent","value_type":"decimal-string"}],"dataset_id":"study.delve-index.tier-skew","description":"Gloss token occurrences attributed to the tier of the matching safe record, compared with how the safe population itself distributes across tiers.","primary_download":false,"rows":[{"cohort":"Tier 1 safe records","record_share_percent":"2.52","row_id":"tier-1","safe_record_count":9943,"token_occurrences":2228560,"token_share_percent":"74.94"},{"cohort":"Tier 2 safe records","record_share_percent":"3.83","row_id":"tier-2","safe_record_count":15100,"token_occurrences":430085,"token_share_percent":"14.46"},{"cohort":"Tier 3 safe records","record_share_percent":"37.43","row_id":"tier-3","safe_record_count":147637,"token_occurrences":251707,"token_share_percent":"8.46"},{"cohort":"Tier 4 safe records","record_share_percent":"0.44","row_id":"tier-4","safe_record_count":1744,"token_occurrences":29018,"token_share_percent":"0.98"},{"cohort":"Tier 5 safe records","record_share_percent":"55.78","row_id":"tier-5","safe_record_count":220021,"token_occurrences":27980,"token_share_percent":"0.94"},{"cohort":"Tokens that are not safe compiled records","record_share_percent":null,"row_id":"not-a-safe-record","safe_record_count":null,"token_occurrences":6562,"token_share_percent":"0.22"}],"title":"Gloss vocabulary versus lexicon tiers"},{"columns":[{"calculation_id":null,"column_id":"row_id","label":"Row ID","role":"dimension","unit":null,"value_type":"text"},{"calculation_id":null,"column_id":"token","label":"Gloss token","role":"dimension","unit":null,"value_type":"text"},{"calculation_id":null,"column_id":"lexicon_tier","label":"Tier of the matching record","role":"dimension","unit":null,"value_type":"integer"},{"calculation_id":"research.count-v1","column_id":"gloss_occurrences","label":"Gloss occurrences","role":"measure","unit":"tokens","value_type":"integer"},{"calculation_id":"research.count-v1","column_id":"documented_headword_count","label":"Documented headwords using it","role":"measure","unit":"records","value_type":"integer"},{"calculation_id":"delve.rate-per-10k-v1","column_id":"rate_per_10k","label":"Occurrences per 10,000 tokens","role":"measure","unit":"per-10k-tokens","value_type":"decimal-string"}],"dataset_id":"study.delve-index.most-used-rarer-band-tokens","description":"A row-limited computed exemplar: gloss tokens whose own compiled record sits in tiers 3-5, ranked by exact gloss occurrences with alphabetical tie-breaks.","primary_download":true,"rows":[{"documented_headword_count":12430,"gloss_occurrences":12562,"lexicon_tier":3,"rate_per_10k":"42.24","row_id":"token-participle","token":"participle"},{"documented_headword_count":5103,"gloss_occurrences":5446,"lexicon_tier":5,"rate_per_10k":"18.31","row_id":"token-s","token":"s"},{"documented_headword_count":4555,"gloss_occurrences":4706,"lexicon_tier":3,"rate_per_10k":"15.82","row_id":"token-is","token":"is"},{"documented_headword_count":3689,"gloss_occurrences":3720,"lexicon_tier":3,"rate_per_10k":"12.51","row_id":"token-genus","token":"genus"},{"documented_headword_count":3156,"gloss_occurrences":3258,"lexicon_tier":4,"rate_per_10k":"10.96","row_id":"token-british","token":"british"},{"documented_headword_count":1780,"gloss_occurrences":1804,"lexicon_tier":4,"rate_per_10k":"6.07","row_id":"token-american","token":"american"},{"documented_headword_count":1693,"gloss_occurrences":1770,"lexicon_tier":4,"rate_per_10k":"5.95","row_id":"token-scottish","token":"scottish"},{"documented_headword_count":1715,"gloss_occurrences":1738,"lexicon_tier":3,"rate_per_10k":"5.84","row_id":"token-dialectal","token":"dialectal"},{"documented_headword_count":1720,"gloss_occurrences":1723,"lexicon_tier":3,"rate_per_10k":"5.79","row_id":"token-superlative","token":"superlative"},{"documented_headword_count":1525,"gloss_occurrences":1565,"lexicon_tier":3,"rate_per_10k":"5.26","row_id":"token-it","token":"it"},{"documented_headword_count":1029,"gloss_occurrences":1044,"lexicon_tier":4,"rate_per_10k":"3.51","row_id":"token-latin","token":"latin"},{"documented_headword_count":814,"gloss_occurrences":821,"lexicon_tier":4,"rate_per_10k":"2.76","row_id":"token-asia","token":"asia"},{"documented_headword_count":798,"gloss_occurrences":805,"lexicon_tier":4,"rate_per_10k":"2.71","row_id":"token-america","token":"america"},{"documented_headword_count":717,"gloss_occurrences":801,"lexicon_tier":3,"rate_per_10k":"2.69","row_id":"token-abbreviation","token":"abbreviation"},{"documented_headword_count":759,"gloss_occurrences":781,"lexicon_tier":4,"rate_per_10k":"2.63","row_id":"token-spanish","token":"spanish"},{"documented_headword_count":562,"gloss_occurrences":717,"lexicon_tier":4,"rate_per_10k":"2.41","row_id":"token-th","token":"th"},{"documented_headword_count":666,"gloss_occurrences":706,"lexicon_tier":4,"rate_per_10k":"2.37","row_id":"token-italian","token":"italian"},{"documented_headword_count":590,"gloss_occurrences":603,"lexicon_tier":4,"rate_per_10k":"2.03","row_id":"token-african","token":"african"},{"documented_headword_count":587,"gloss_occurrences":596,"lexicon_tier":4,"rate_per_10k":"2.00","row_id":"token-africa","token":"africa"},{"documented_headword_count":499,"gloss_occurrences":535,"lexicon_tier":3,"rate_per_10k":"1.80","row_id":"token-first","token":"first"},{"documented_headword_count":465,"gloss_occurrences":479,"lexicon_tier":4,"rate_per_10k":"1.61","row_id":"token-australian","token":"australian"},{"documented_headword_count":457,"gloss_occurrences":464,"lexicon_tier":3,"rate_per_10k":"1.56","row_id":"token-ornamental","token":"ornamental"},{"documented_headword_count":432,"gloss_occurrences":438,"lexicon_tier":3,"rate_per_10k":"1.47","row_id":"token-chiefly","token":"chiefly"},{"documented_headword_count":437,"gloss_occurrences":437,"lexicon_tier":3,"rate_per_10k":"1.47","row_id":"token-notably","token":"notably"},{"documented_headword_count":429,"gloss_occurrences":435,"lexicon_tier":4,"rate_per_10k":"1.46","row_id":"token-asian","token":"asian"}],"title":"Most-used rarer-band gloss vocabulary"},{"columns":[{"calculation_id":null,"column_id":"row_id","label":"Row ID","role":"dimension","unit":null,"value_type":"text"},{"calculation_id":null,"column_id":"marker","label":"Marker","role":"dimension","unit":null,"value_type":"text"},{"calculation_id":null,"column_id":"lexicon_tier","label":"Tier of the matching record","role":"dimension","unit":null,"value_type":"integer"},{"calculation_id":"delve.style-marker-count-v1","column_id":"gloss_occurrences","label":"Gloss occurrences","role":"measure","unit":"tokens","value_type":"integer"},{"calculation_id":"delve.style-marker-count-v1","column_id":"documented_headword_count","label":"Distinct documented headwords using it","role":"measure","unit":"records","value_type":"integer"},{"calculation_id":"delve.rate-per-10k-v1","column_id":"rate_per_10k","label":"Occurrences per 10,000 tokens","role":"measure","unit":"per-10k-tokens","value_type":"decimal-string"}],"dataset_id":"study.delve-index.style-markers","description":"Exact gloss counts for a fixed, version-controlled list of terms associated with machine-written explanatory prose. List membership is a declared heuristic.","primary_download":false,"rows":[{"documented_headword_count":26,"gloss_occurrences":29,"lexicon_tier":2,"marker":"boast","rate_per_10k":"0.10","row_id":"marker-boast"},{"documented_headword_count":12,"gloss_occurrences":12,"lexicon_tier":1,"marker":"crucial","rate_per_10k":"0.04","row_id":"marker-crucial"},{"documented_headword_count":6,"gloss_occurrences":6,"lexicon_tier":2,"marker":"delve","rate_per_10k":"0.02","row_id":"marker-delve"},{"documented_headword_count":50,"gloss_occurrences":50,"lexicon_tier":2,"marker":"emphasize","rate_per_10k":"0.17","row_id":"marker-emphasize"},{"documented_headword_count":14,"gloss_occurrences":14,"lexicon_tier":3,"marker":"encompass","rate_per_10k":"0.05","row_id":"marker-encompass"},{"documented_headword_count":20,"gloss_occurrences":21,"lexicon_tier":2,"marker":"evoke","rate_per_10k":"0.07","row_id":"marker-evoke"},{"documented_headword_count":24,"gloss_occurrences":24,"lexicon_tier":2,"marker":"facilitate","rate_per_10k":"0.08","row_id":"marker-facilitate"},{"documented_headword_count":13,"gloss_occurrences":15,"lexicon_tier":1,"marker":"foster","rate_per_10k":"0.05","row_id":"marker-foster"},{"documented_headword_count":54,"gloss_occurrences":56,"lexicon_tier":2,"marker":"intricate","rate_per_10k":"0.19","row_id":"marker-intricate"},{"documented_headword_count":6,"gloss_occurrences":7,"lexicon_tier":1,"marker":"leverage","rate_per_10k":"0.02","row_id":"marker-leverage"},{"documented_headword_count":13,"gloss_occurrences":13,"lexicon_tier":2,"marker":"meticulous","rate_per_10k":"0.04","row_id":"marker-meticulous"},{"documented_headword_count":0,"gloss_occurrences":0,"lexicon_tier":3,"marker":"multifaceted","rate_per_10k":"0.00","row_id":"marker-multifaceted"},{"documented_headword_count":1,"gloss_occurrences":1,"lexicon_tier":2,"marker":"myriad","rate_per_10k":"0.00","row_id":"marker-myriad"},{"documented_headword_count":437,"gloss_occurrences":437,"lexicon_tier":3,"marker":"notably","rate_per_10k":"1.47","row_id":"marker-notably"},{"documented_headword_count":1,"gloss_occurrences":1,"lexicon_tier":3,"marker":"nuanced","rate_per_10k":"0.00","row_id":"marker-nuanced"},{"documented_headword_count":2,"gloss_occurrences":2,"lexicon_tier":2,"marker":"pivotal","rate_per_10k":"0.01","row_id":"marker-pivotal"},{"documented_headword_count":100,"gloss_occurrences":101,"lexicon_tier":1,"marker":"realm","rate_per_10k":"0.34","row_id":"marker-realm"},{"documented_headword_count":25,"gloss_occurrences":25,"lexicon_tier":2,"marker":"robust","rate_per_10k":"0.08","row_id":"marker-robust"},{"documented_headword_count":3,"gloss_occurrences":4,"lexicon_tier":2,"marker":"showcase","rate_per_10k":"0.01","row_id":"marker-showcase"},{"documented_headword_count":13,"gloss_occurrences":15,"lexicon_tier":2,"marker":"tapestry","rate_per_10k":"0.05","row_id":"marker-tapestry"},{"documented_headword_count":5,"gloss_occurrences":7,"lexicon_tier":3,"marker":"underscore","rate_per_10k":"0.02","row_id":"marker-underscore"},{"documented_headword_count":5,"gloss_occurrences":6,"lexicon_tier":2,"marker":"vibrant","rate_per_10k":"0.02","row_id":"marker-vibrant"},{"documented_headword_count":826,"gloss_occurrences":846,"lexicon_tier":null,"marker":"All declared markers combined","rate_per_10k":"2.84","row_id":"all-markers-combined"}],"title":"Declared style-marker audit"}],"description":"A frequency self-audit of the first-party machine-written definition corpus against the compiled lexicon: gloss coverage, tier skew, most-used rarer-band vocabulary, and a declared style-marker audit with methodology stated first.","downloads":[{"contains":["study.delve-index.population","study.delve-index.corpus","study.delve-index.gloss-coverage","study.delve-index.tier-skew","study.delve-index.most-used-rarer-band-tokens","study.delve-index.style-markers"],"filename":"delve-index.csv","media_type":"text/csv","sha256":"e329b02c4d9c222eda251e227b39a8942e9c49ff1a9c3876171da20b2fbadcde","size_bytes":132541,"status":"available-review-pending"}],"exclusions":[{"criterion_id":"delve-index.product-policy","description":"Exclude spellings matched by the exact first-party content-safety policy; excluded spellings can never appear as joined tokens."},{"criterion_id":"delve-index.non-alphabetic-runs","description":"Tokens are exact lowercase a-z runs; digits, punctuation, and whitespace delimit tokens and are never counted."}],"facts":[{"calculation_id":"research.count-v1","column_id":"record_count","dataset_id":"study.delve-index.population","denominator":null,"display_value":"394,445","fact_id":"study.delve-index.analysis-record-count","label":"Analysis population","numerator":null,"row_id":"analysis","unit":"records","value":394445},{"calculation_id":"research.share-v1","column_id":"share_percent","dataset_id":"study.delve-index.gloss-coverage","denominator":394445,"display_value":"59.98%","fact_id":"study.delve-index.documented-share","label":"Safe records documented by first-party glosses","numerator":236598,"row_id":"documented","unit":"percent","value":"59.98"},{"calculation_id":"delve.style-marker-count-v1","column_id":"gloss_occurrences","dataset_id":"study.delve-index.style-markers","denominator":null,"display_value":"846","fact_id":"study.delve-index.marker-occurrences","label":"Combined occurrences of the declared style markers","numerator":null,"row_id":"all-markers-combined","unit":"tokens","value":846},{"calculation_id":"delve.token-leader-v1","column_id":"token","dataset_id":"study.delve-index.most-used-rarer-band-tokens","denominator":null,"display_value":"participle","fact_id":"study.delve-index.top-rarer-band-token","label":"Most-used rarer-band gloss token","numerator":null,"row_id":"token-participle","unit":"token","value":"participle"}],"inclusion_criteria":[{"criterion_id":"delve-index.first-party-glosses","description":"Include every first-party definition gloss attached to a safe compiled record in this release."},{"criterion_id":"delve-index.safe-lexicon-join","description":"Join each distinct gloss token to the safe compiled record with the same normalized spelling, where one exists."}],"initial_cohort":false,"lexical_build_id":"647a998abcbf65af6a16c86cb8460cb4a88761a94544468e9c9e992208fc327c","lexical_manifest_sha256":"46665edef6a77a00080debaa2209f630f4d220ad2888055fa730731f93dc0561","lexical_release_timestamp":"2026-07-26T00:00:00+00:00","limitations":["The glosses are first-party machine-generated text; every rate describes that corpus's writing style, not English usage.","Token counting ignores senses, multi-word phrases, and grammar; a marker can be the only correct word in its context.","Splitting on non-letters fragments possessives and abbreviations, so single-letter tokens such as 's' occur.","The style-marker list is a fixed editorial heuristic recorded in version control.","Tier joins are build-specific ranking bands, not universal usage claims."],"narrative":[{"block_id":"study.delve-index.method","fact_ids":[],"heading":"Methodology first: this corpus audits itself","interpretation_kind":"methodological","paragraphs":["The definition glosses are first-party, machine-generated text governed by this product's review policy; inclusion in the build is not word-level editorial approval. This study counts exact lowercase a-z token runs in those glosses and joins each token to its own compiled record, so every rate describes the corpus's writing style, not English.","The style-marker list is a fixed editorial heuristic recorded in version control. Appearing on it is not evidence about any individual gloss."]},{"block_id":"study.delve-index.result","fact_ids":["study.delve-index.documented-share","study.delve-index.marker-occurrences","study.delve-index.top-rarer-band-token"],"heading":null,"interpretation_kind":"computed","paragraphs":["First-party glosses document 59.98% of safe records; the 22 declared style markers occur 846 times in 2,973,912 gloss tokens."]},{"block_id":"study.delve-index.scope","fact_ids":[],"heading":"What the audit cannot claim","interpretation_kind":"methodological","paragraphs":["Token counting ignores senses, phrases, and grammar; a marker can be the only correct word in context. The comparison population is this build's safe compiled lexicon, so shares are build-specific and do not measure general English usage or the style of any other text."]}],"output_license":"No reuse licence asserted; source-rights review pending","output_license_url":"/about/data","population_sha256":"e9d14e35f548f80eb3d5dd29a14825521acac57214cb72badbbd893f648d05f6","publication_status":"published","related_searches":[{"label":"Look up a word's definition","relevance":"Read the first-party glosses this audit measures.","url":"/words"}],"reproducibility_command":"make studies_reproduce STUDY=delve-index","research_question":"Which tier 3-5 compiled tokens occur most often in the first-party definition corpus, and how often do declared machine-prose style markers appear?","review":{"approved_analysis_contract_digest":"cf75d7ed5dbb8022756ef7a00652f3f3e991013061ba1d4e2b967139d026ee03","approved_content_artifact_digest":"477542693badcafd96a49a013917cb986d413ee48d738561b13d04fa8f330416","approved_content_version":"1.0.1","approved_data_version":"2026.07-lexical.6","approved_lexical_build_id":"647a998abcbf65af6a16c86cb8460cb4a88761a94544468e9c9e992208fc327c","approved_result_sha256":"56ff1c5befbde56bcead61ffd1a4e6307d6834ceb2c9d084bb46991a9ac68687","editorial_approval":"TWI-v1.4.0-editorial-review-2026-08-02-r1","editorial_status":"approved","factual_approval":null,"factual_status":"machine-validated","legal_approval":null,"legal_status":"pending","reevaluate_on":"2026-10-17","reviewed_at":"2026-08-02","reviewer":"Matthew Iles","search_approval":null,"search_eligibility_reason":"The self-audit remains noindex pending source-rights review and separate search approval.","search_eligible":false},"schema_version":1,"slug":"delve-index","sources":[{"artifact_sha256":"3ed0c94610d8bcf7c11bbb49c56aa49c7234d32b66824df91f554169e572da48","attribution":"dwyl/english-words; the repository carries an Unlicense notice, while its pinned README attributes upstream copyright to Infochimps","establishes":["source-membership","surface-spelling-as-listed"],"legal_review_status":"pending","licence_name":"Unlicense notice; upstream dataset rights unresolved","licence_url":"https://raw.githubusercontent.com/dwyl/english-words/8179fe68775df3f553ef19520db065228e65d1d3/LICENSE.md","limitations":["The pinned repository licence is the Unlicense, but the pinned README says copyright in the extracted source list remains with Infochimps. Human legal review of the upstream rights chain is required.","Do not assert a redistribution right until the Infochimps-to-dwyl rights chain has been reviewed by a human.","Dialect or scope: unspecified"],"redistribution_status":"pending","source_class":"broad-orthographic-list","source_id":"dwyl-english-words","source_version":"git:8179fe68775df3f553ef19520db065228e65d1d3","url":"https://raw.githubusercontent.com/dwyl/english-words/8179fe68775df3f553ef19520db065228e65d1d3/words_alpha.txt"},{"artifact_sha256":"d3fbe8485022088fcf527edcde2fbdc18b4bbc141ac58123c9adb462e086eaf7","attribution":"ENABLE (Enhanced North American Benchmark Lexicon); exact reuse terms are pending legal review","establishes":["source-membership","surface-spelling-as-listed"],"legal_review_status":"pending","licence_name":"Licence terms not stated by the source","licence_url":"https://www.wordgamedictionary.com/enable/","limitations":["The downloaded artifact is content-pinned by SHA-256, but it contains no licence notice and the configured source page does not state reuse terms.","The artifact is reproducibly pinned but no affirmative reuse grant has been located; redistribution approval remains pending.","Dialect or scope: North-American-oriented; exact edition metadata unavailable"],"redistribution_status":"pending","source_class":"historical-game-word-list","source_id":"enable","source_version":"artifact-sha256:d3fbe8485022088fcf527edcde2fbdc18b4bbc141ac58123c9adb462e086eaf7","url":"https://www.wordgamedictionary.com/enable/download/enable.txt"},{"artifact_sha256":"ee2d83651fbb91642bbed2bd30ead404c2cfbdfece01dacf284af6ea47795811","attribution":"first20hours/google-10000-english, derived from the Google Web Trillion Word Corpus and Peter Norvig's compilation","establishes":["source-membership","source-order"],"legal_review_status":"pending","licence_name":"No explicit licence grant in pinned repository metadata","licence_url":"https://raw.githubusercontent.com/first20hours/google-10000-english/bdf4c221bc120b0b7f6c3f1eff1cc1abb975f8d8/LICENSE.md","limitations":["The exact pinned LICENSE.md describes provenance but contains no explicit public-domain dedication or licence grant. Human review is required.","The pinned repository documents provenance but does not supply an affirmative licence grant; do not describe it as public domain.","Dialect or scope: USA no-swears variant"],"redistribution_status":"pending","source_class":"ranked-web-corpus-derived-list","source_id":"google-10000-english","source_version":"git:bdf4c221bc120b0b7f6c3f1eff1cc1abb975f8d8","url":"https://raw.githubusercontent.com/first20hours/google-10000-english/bdf4c221bc120b0b7f6c3f1eff1cc1abb975f8d8/google-10000-english-usa-no-swears.txt"},{"artifact_sha256":"5351ff405b1126ef555791dd4d9798a48e3e9a501a9fc481a9da957752cfb458","attribution":"FrequencyWords content derived from OpenSubtitles, licensed CC BY-SA 4.0","establishes":["source-token-observation","source-token-count"],"legal_review_status":"pending","licence_name":"CC BY-SA 4.0 (content)","licence_url":"https://raw.githubusercontent.com/hermitdave/FrequencyWords/525f9b560de45753a5ea01069454e72e9aa541c6/README.md","limitations":["The exact pinned README states MIT for code and CC BY-SA 4.0 for content. Human review is required for attribution and ShareAlike obligations on compiled and derived data.","Content is declared CC BY-SA 4.0. Attribution and ShareAlike treatment of the compiled frequency evidence and derived tiers require human approval.","Dialect or scope: unspecified"],"redistribution_status":"pending","source_class":"subtitle-derived-token-frequency","source_id":"frequencywords-en-2018","source_version":"git:525f9b560de45753a5ea01069454e72e9aa541c6; corpus-release:2018","url":"https://raw.githubusercontent.com/hermitdave/FrequencyWords/525f9b560de45753a5ea01069454e72e9aa541c6/content/2018/en/en_50k.txt"},{"artifact_sha256":"ef70bc128a3ae67c23c5a4e31ec6b0ef3780e94881faa3adf9b73c2f7f684b4d","attribution":"The Word Index first-party content-safety exclusion policy","establishes":["content_filter","search_results","result_counts"],"legal_review_status":"pending","licence_name":"Reuse terms not yet approved","licence_url":null,"limitations":["This bounded policy list cannot determine whether every usage is offensive or benign."],"redistribution_status":"pending","source_class":"first-party-content-safety-policy","source_id":"offensive-term-policy","source_version":"2026.07-review-pending.2","url":"/about/data"}],"study_build_id":"f59387dbfd58d904841d0c2f779a4f1d21a9f86f94e62fd57643be45a544b517","study_id":"study.delve-index","study_version":"1.0.1","tables":[{"caption":"The exact population used by this version of the analysis.","column_ids":["stage","record_count"],"dataset_id":"study.delve-index.population","heading":"Population and exclusions","note":null,"row_ids":["raw","policy-excluded","safe","analysis"],"table_id":"study.delve-index.population-table"},{"caption":"Corpus sizes are exact counts over the sealed first-party glosses.","column_ids":["metric","observation_count"],"dataset_id":"study.delve-index.corpus","heading":"What was counted","note":null,"row_ids":["documented-headwords","definition-glosses","gloss-token-occurrences","distinct-gloss-tokens","tokens-matching-safe-records"],"table_id":"study.delve-index.corpus-table"},{"caption":"Coverage describes this build's first-party corpus only.","column_ids":["coverage_class","record_count","share_percent"],"dataset_id":"study.delve-index.gloss-coverage","heading":"Gloss coverage","note":null,"row_ids":["documented","undocumented"],"table_id":"study.delve-index.coverage-table"},{"caption":"Token shares and record shares use different exact denominators, both named in the dataset.","column_ids":["cohort","token_occurrences","token_share_percent","safe_record_count","record_share_percent"],"dataset_id":"study.delve-index.tier-skew","heading":"Where gloss vocabulary sits in the lexicon","note":null,"row_ids":["tier-1","tier-2","tier-3","tier-4","tier-5","not-a-safe-record"],"table_id":"study.delve-index.tier-skew-table"},{"caption":"A computed exemplar limited to 25 rows; policy-excluded spellings can never appear.","column_ids":["token","lexicon_tier","gloss_occurrences","documented_headword_count","rate_per_10k"],"dataset_id":"study.delve-index.most-used-rarer-band-tokens","heading":"Most-used tier 3-5 vocabulary in the glosses","note":null,"row_ids":["token-participle","token-s","token-is","token-genus","token-british","token-american","token-scottish","token-dialectal","token-superlative","token-it","token-latin","token-asia","token-america","token-abbreviation","token-spanish","token-th","token-italian","token-african","token-africa","token-first","token-australian","token-ornamental","token-chiefly","token-notably","token-asian"],"table_id":"study.delve-index.most-used-rarer-band-table"},{"caption":"Zero counts are published so the audit cannot cherry-pick.","column_ids":["marker","lexicon_tier","gloss_occurrences","documented_headword_count","rate_per_10k"],"dataset_id":"study.delve-index.style-markers","heading":"The declared style-marker list, in full","note":null,"row_ids":["marker-boast","marker-crucial","marker-delve","marker-emphasize","marker-encompass","marker-evoke","marker-facilitate","marker-foster","marker-intricate","marker-leverage","marker-meticulous","marker-multifaceted","marker-myriad","marker-notably","marker-nuanced","marker-pivotal","marker-realm","marker-robust","marker-showcase","marker-tapestry","marker-underscore","marker-vibrant","all-markers-combined"],"table_id":"study.delve-index.marker-table"}],"title":"The Delve Index: a self-audit of our machine-written definition corpus","transformation_steps":[{"description":"Casefold every first-party gloss and extract exact lowercase a-z token runs.","deterministic":true,"limitations":["Tokenization has no linguistic model; hyphenated and possessive forms split."],"transformation_id":"delve-index.tokenize-v1"},{"description":"Join each distinct token to the safe compiled record with the same spelling and attribute occurrences to that record's tier.","deterministic":true,"limitations":["Tokens without a matching safe record are reported in a single unjoined cohort."],"transformation_id":"delve-index.join-v1"},{"description":"Count occurrences of the fixed declared style-marker list, count the distinct documented headwords containing any marker, and rank tier 3-5 tokens by exact occurrences, limited to 25 rows.","deterministic":true,"limitations":[],"transformation_id":"delve-index.audit-v1"}],"update_history":[{"data_version":"2026.07-lexical.4","effective_date":"2026-07-19","result_sha256":"01782df5414be32c81f0dca3866968456dc43d6b221b4e27ec35d4df312b1d30","review_status":"machine-validated","reviewer":null,"study_version":"1.0.0","summary":"Initial reproducible definition-corpus self-audit."},{"data_version":"2026.07-lexical.5","effective_date":"2026-07-25","result_sha256":"01782df5414be32c81f0dca3866968456dc43d6b221b4e27ec35d4df312b1d30","review_status":"machine-validated","reviewer":null,"study_version":"1.0.0","summary":"Rebound byte-identical results to the 2026.07-lexical.5 data release; no numeric changes."},{"data_version":"2026.07-lexical.6","effective_date":"2026-07-26","result_sha256":"01782df5414be32c81f0dca3866968456dc43d6b221b4e27ec35d4df312b1d30","review_status":"machine-validated","reviewer":null,"study_version":"1.0.0","summary":"Rebound byte-identical results to the 2026.07-lexical.6 data release; no numeric changes."},{"data_version":"2026.07-lexical.6","effective_date":"2026-08-01","result_sha256":"56ff1c5befbde56bcead61ffd1a4e6307d6834ceb2c9d084bb46991a9ac68687","review_status":"machine-validated","reviewer":null,"study_version":"1.0.1","summary":"Corrected the combined marker headword count to a distinct union and reframed the tier 3-5 token table as frequency, not overuse."}],"user_relevance":"Documents, with exact counts, which tier 3-5 tokens recur most often in this site's machine-written definitions and how frequently the declared style markers appear."}
