{"analysis_contract_sha256":"214ba6800412c85e2bbeb61d6e5c006eb2dbb933ab5082c4f3323745c4ef70d3","analysis_id":"analysis.spelling-difficulty-v1","calculations":[{"calculation_id":"research.count-v1","confidence":"deterministic","description":"Count records in each declared population stage and score cohort.","formula":"count(records)","limitations":[],"source_fields":["normalized_spelling","pronunciation_variants","frequency_tier"]},{"calculation_id":"research.share-v1","confidence":"deterministic","description":"Calculate a cohort's exact share of its named denominator.","formula":"100 * subset_count / cohort_count","limitations":["Shares describe the covered compiled population, not running text."],"source_fields":["normalized_spelling","pronunciation_variants","frequency_tier"]},{"calculation_id":"research.ratio-v1","confidence":"deterministic","description":"Divide letter count by primary-variant phoneme count, half-up to 2 places.","formula":"letters / primary_phonemes","limitations":["Phoneme counts follow the pinned CMUdict transcription conventions."],"source_fields":["normalized_spelling","pronunciation_variants"]},{"calculation_id":"research.maximum-v1","confidence":"deterministic","description":"Select the highest heuristic score within a declared cohort.","formula":"max(formula_score)","limitations":[],"source_fields":["normalized_spelling","pronunciation_variants"]},{"calculation_id":"orthography.formula-component-v1","confidence":"heuristic","description":"Compute the three declared formula components: letter-phoneme count surplus, vowel-sequence points from a longest-first scan, and doubled-letter pairs.","formula":"surplus = max(0, letters - primary_phonemes); points = sum(declared grapheme weights, longest-first scan); pairs = count(adjacent identical letters)","limitations":["Components are declared heuristics, not measured reader or speller errors."],"source_fields":["normalized_spelling","pronunciation_variants"]},{"calculation_id":"orthography.formula-score-v1","confidence":"heuristic","description":"Combine the declared components with fixed version-controlled weights.","formula":"3 * max(0, letters - primary_phonemes) + vowel_grapheme_points + 2 * doubled_letter_pairs","limitations":["Weights are fixed editorial choices; the score is not fitted to any error data."],"source_fields":["normalized_spelling","pronunciation_variants"]}],"charts":[{"accessible_summary":"Exact band counts and shares appear in the adjacent table.","category_column_id":"band","chart_id":"study.spelling-difficulty.bands-chart","chart_type":"grouped-bar","dataset_id":"study.spelling-difficulty.bands","heading":"Covered records per formula-score band","series":[{"label":"All covered","value_column_id":"all_record_count"},{"label":"Tiers 1-2","value_column_id":"common_record_count"}],"table_id":"study.spelling-difficulty.bands-table"}],"citations":[{"citation_id":"citation.lexical-sources","citation_kind":"dataset","data_version":"2026.07-lexical.6","publisher":"The Word Index","source_id":null,"title":"The Word Index pinned lexical source manifest","url":"/about/data","version":"2026.07-lexical.6"},{"citation_id":"citation.cmudict","citation_kind":"dataset","data_version":null,"publisher":"Carnegie Mellon Speech Group","source_id":"cmudict","title":"CMU Pronouncing Dictionary","url":"https://raw.githubusercontent.com/cmusphinx/cmudict/74790861f652b15e4ac49015a90074ad62a27690/cmudict.dict","version":"git:74790861f652b15e4ac49015a90074ad62a27690"},{"citation_id":"citation.frequencywords","citation_kind":"dataset","data_version":null,"publisher":"FrequencyWords","source_id":"frequencywords-en-2018","title":"FrequencyWords English 2018","url":"https://raw.githubusercontent.com/hermitdave/FrequencyWords/525f9b560de45753a5ea01069454e72e9aa541c6/content/2018/en/en_50k.txt","version":"git:525f9b560de45753a5ea01069454e72e9aa541c6; corpus-release:2018"},{"citation_id":"citation.content-policy","citation_kind":"policy","data_version":null,"publisher":"The Word Index","source_id":"offensive-term-policy","title":"The Word Index content-safety exclusion policy","url":"/about/data","version":"2026.07-review-pending.2"}],"concise_result":"31.55% of tier 1-2 covered records score at least 6 points under the declared formula; the highest tier 1-2 score is 22.","content_artifact_digest":"477542693badcafd96a49a013917cb986d413ee48d738561b13d04fa8f330416","data_version":"2026.07-lexical.6","datasets":[{"columns":[{"calculation_id":null,"column_id":"row_id","label":"Row ID","role":"dimension","unit":null,"value_type":"text"},{"calculation_id":null,"column_id":"stage","label":"Population stage","role":"dimension","unit":null,"value_type":"text"},{"calculation_id":"research.count-v1","column_id":"record_count","label":"Records","role":"measure","unit":"records","value_type":"integer"}],"dataset_id":"study.spelling-difficulty.population","description":"Exact record counts before and after the declared product-policy and study filters.","primary_download":false,"rows":[{"record_count":394595,"row_id":"raw","stage":"Raw lexical ledger"},{"record_count":150,"row_id":"policy-excluded","stage":"Excluded by product policy"},{"record_count":394445,"row_id":"safe","stage":"Safe normalized-spelling records"},{"record_count":59003,"row_id":"analysis","stage":"Records meeting this study's inclusion criteria"}],"title":"Study population"},{"columns":[{"calculation_id":null,"column_id":"row_id","label":"Row ID","role":"dimension","unit":null,"value_type":"text"},{"calculation_id":null,"column_id":"metric","label":"Metric","role":"dimension","unit":null,"value_type":"text"},{"calculation_id":"research.count-v1","column_id":"record_count","label":"Records","role":"measure","unit":"records","value_type":"integer"},{"calculation_id":"research.count-v1","column_id":"denominator","label":"Denominator","role":"measure","unit":"records","value_type":"integer"},{"calculation_id":"research.share-v1","column_id":"share_percent","label":"Share","role":"measure","unit":"percent","value_type":"decimal-string"},{"calculation_id":"research.maximum-v1","column_id":"score_value","label":"Score","role":"measure","unit":"points","value_type":"integer"}],"dataset_id":"study.spelling-difficulty.summary","description":"Headline counts and maxima derived from the declared scoring formula.","primary_download":false,"rows":[{"denominator":59003,"metric":"Covered records in tiers 1-2","record_count":24183,"row_id":"covered-tier-1-2","score_value":null,"share_percent":"40.99"},{"denominator":24183,"metric":"Tier 1-2 records scoring at least 6 points","record_count":7630,"row_id":"tier-1-2-at-least-6","score_value":null,"share_percent":"31.55"},{"denominator":null,"metric":"Highest heuristic score among tier 1-2 covered records","record_count":null,"row_id":"highest-tier-1-2-score","score_value":22,"share_percent":null},{"denominator":null,"metric":"Highest heuristic score among all covered records","record_count":null,"row_id":"highest-all-score","score_value":24,"share_percent":null}],"title":"Formula-score summary"},{"columns":[{"calculation_id":null,"column_id":"row_id","label":"Row ID","role":"dimension","unit":null,"value_type":"text"},{"calculation_id":null,"column_id":"band","label":"Formula-score band","role":"dimension","unit":null,"value_type":"text"},{"calculation_id":null,"column_id":"score_range","label":"Score range","role":"dimension","unit":null,"value_type":"text"},{"calculation_id":"research.count-v1","column_id":"all_record_count","label":"Covered records","role":"measure","unit":"records","value_type":"integer"},{"calculation_id":"research.share-v1","column_id":"all_share_percent","label":"Covered share","role":"measure","unit":"percent","value_type":"decimal-string"},{"calculation_id":"research.count-v1","column_id":"common_record_count","label":"Tier 1-2 records","role":"measure","unit":"records","value_type":"integer"},{"calculation_id":"research.share-v1","column_id":"common_share_percent","label":"Tier 1-2 share","role":"measure","unit":"percent","value_type":"decimal-string"}],"dataset_id":"study.spelling-difficulty.bands","description":"Covered records grouped by total points under the declared formula, for the full covered cohort and the tier 1-2 cohort.","primary_download":true,"rows":[{"all_record_count":15129,"all_share_percent":"25.64","band":"0","common_record_count":6105,"common_share_percent":"25.25","row_id":"band-0","score_range":"0 to 0"},{"all_record_count":931,"all_share_percent":"1.58","band":"1-2","common_record_count":316,"common_share_percent":"1.31","row_id":"band-1-2","score_range":"1 to 2"},{"all_record_count":23733,"all_share_percent":"40.22","band":"3-5","common_record_count":10132,"common_share_percent":"41.90","row_id":"band-3-5","score_range":"3 to 5"},{"all_record_count":15786,"all_share_percent":"26.75","band":"6-9","common_record_count":6393,"common_share_percent":"26.44","row_id":"band-6-9","score_range":"6 to 9"},{"all_record_count":3424,"all_share_percent":"5.80","band":"10-plus","common_record_count":1237,"common_share_percent":"5.12","row_id":"band-10-plus","score_range":"10 and above"}],"title":"Declared formula-score bands"},{"columns":[{"calculation_id":null,"column_id":"row_id","label":"Row ID","role":"dimension","unit":null,"value_type":"text"},{"calculation_id":null,"column_id":"grapheme","label":"Grapheme","role":"dimension","unit":null,"value_type":"text"},{"calculation_id":null,"column_id":"weight","label":"Declared points","role":"dimension","unit":null,"value_type":"integer"},{"calculation_id":"research.count-v1","column_id":"record_count","label":"Covered records containing it","role":"measure","unit":"records","value_type":"integer"},{"calculation_id":"research.share-v1","column_id":"share_percent","label":"Share of covered records","role":"measure","unit":"percent","value_type":"decimal-string"}],"dataset_id":"study.spelling-difficulty.graphemes","description":"How many covered records match each declared grapheme at least once under the deterministic longest-first scan.","primary_download":false,"rows":[{"grapheme":"ae","record_count":117,"row_id":"grapheme-ae","share_percent":"0.20","weight":1},{"grapheme":"aigh","record_count":12,"row_id":"grapheme-aigh","share_percent":"0.02","weight":3},{"grapheme":"augh","record_count":39,"row_id":"grapheme-augh","share_percent":"0.07","weight":3},{"grapheme":"ea","record_count":2225,"row_id":"grapheme-ea","share_percent":"3.77","weight":1},{"grapheme":"eau","record_count":47,"row_id":"grapheme-eau","share_percent":"0.08","weight":2},{"grapheme":"ei","record_count":423,"row_id":"grapheme-ei","share_percent":"0.72","weight":1},{"grapheme":"eigh","record_count":75,"row_id":"grapheme-eigh","share_percent":"0.13","weight":3},{"grapheme":"eou","record_count":56,"row_id":"grapheme-eou","share_percent":"0.09","weight":2},{"grapheme":"ie","record_count":2018,"row_id":"grapheme-ie","share_percent":"3.42","weight":1},{"grapheme":"ieu","record_count":9,"row_id":"grapheme-ieu","share_percent":"0.02","weight":2},{"grapheme":"igh","record_count":312,"row_id":"grapheme-igh","share_percent":"0.53","weight":2},{"grapheme":"iou","record_count":197,"row_id":"grapheme-iou","share_percent":"0.33","weight":2},{"grapheme":"oe","record_count":210,"row_id":"grapheme-oe","share_percent":"0.36","weight":1},{"grapheme":"ou","record_count":1810,"row_id":"grapheme-ou","share_percent":"3.07","weight":1},{"grapheme":"ough","record_count":103,"row_id":"grapheme-ough","share_percent":"0.17","weight":3},{"grapheme":"ue","record_count":571,"row_id":"grapheme-ue","share_percent":"0.97","weight":1},{"grapheme":"ui","record_count":545,"row_id":"grapheme-ui","share_percent":"0.92","weight":1}],"title":"Declared vowel sequences"},{"columns":[{"calculation_id":null,"column_id":"row_id","label":"Row ID","role":"dimension","unit":null,"value_type":"text"},{"calculation_id":null,"column_id":"rank","label":"Rank","role":"dimension","unit":null,"value_type":"integer"},{"calculation_id":null,"column_id":"spelling","label":"Normalized spelling","role":"dimension","unit":null,"value_type":"text"},{"calculation_id":null,"column_id":"tier","label":"Tier","role":"dimension","unit":null,"value_type":"integer"},{"calculation_id":null,"column_id":"length","label":"Letters","role":"dimension","unit":null,"value_type":"integer"},{"calculation_id":null,"column_id":"phoneme_count","label":"Primary-variant phonemes","role":"dimension","unit":null,"value_type":"integer"},{"calculation_id":"research.ratio-v1","column_id":"letters_per_phoneme","label":"Letters per phoneme","role":"measure","unit":"ratio","value_type":"decimal-string"},{"calculation_id":"orthography.formula-component-v1","column_id":"letter_phoneme_count_surplus","label":"Letter-phoneme count surplus","role":"measure","unit":"letters","value_type":"integer"},{"calculation_id":"orthography.formula-component-v1","column_id":"vowel_sequence_points","label":"Declared vowel-sequence points","role":"measure","unit":"points","value_type":"integer"},{"calculation_id":"orthography.formula-component-v1","column_id":"doubled_letter_count","label":"Doubled-letter pairs","role":"measure","unit":"pairs","value_type":"integer"},{"calculation_id":"orthography.formula-score-v1","column_id":"formula_score","label":"Formula score","role":"measure","unit":"points","value_type":"integer"}],"dataset_id":"study.spelling-difficulty.highest-scoring-tier-1-2","description":"A row-limited computed exemplar of the highest-scoring tier 1-2 covered records. It is a heuristic ranking for this build, not a source list or a usage claim.","primary_download":false,"rows":[{"doubled_letter_count":0,"formula_score":22,"length":12,"letter_phoneme_count_surplus":6,"letters_per_phoneme":"2.00","phoneme_count":6,"rank":1,"row_id":"rank-1","spelling":"neighbouring","tier":2,"vowel_sequence_points":4},{"doubled_letter_count":0,"formula_score":22,"length":14,"letter_phoneme_count_surplus":6,"letters_per_phoneme":"1.75","phoneme_count":8,"rank":2,"row_id":"rank-2","spelling":"slaughterhouse","tier":2,"vowel_sequence_points":4},{"doubled_letter_count":1,"formula_score":21,"length":13,"letter_phoneme_count_surplus":5,"letters_per_phoneme":"1.63","phoneme_count":8,"rank":3,"row_id":"rank-3","spelling":"righteousness","tier":2,"vowel_sequence_points":4},{"doubled_letter_count":1,"formula_score":20,"length":13,"letter_phoneme_count_surplus":5,"letters_per_phoneme":"1.63","phoneme_count":8,"rank":4,"row_id":"rank-4","spelling":"granddaughter","tier":1,"vowel_sequence_points":3},{"doubled_letter_count":0,"formula_score":20,"length":11,"letter_phoneme_count_surplus":5,"letters_per_phoneme":"1.83","phoneme_count":6,"rank":5,"row_id":"rank-5","spelling":"lightweight","tier":2,"vowel_sequence_points":5},{"doubled_letter_count":1,"formula_score":20,"length":12,"letter_phoneme_count_surplus":5,"letters_per_phoneme":"1.71","phoneme_count":7,"rank":6,"row_id":"rank-6","spelling":"neighborhood","tier":1,"vowel_sequence_points":3},{"doubled_letter_count":1,"formula_score":20,"length":13,"letter_phoneme_count_surplus":5,"letters_per_phoneme":"1.63","phoneme_count":8,"rank":7,"row_id":"rank-7","spelling":"neighborhoods","tier":2,"vowel_sequence_points":3},{"doubled_letter_count":1,"formula_score":20,"length":11,"letter_phoneme_count_surplus":5,"letters_per_phoneme":"1.83","phoneme_count":6,"rank":8,"row_id":"rank-8","spelling":"thoughtless","tier":2,"vowel_sequence_points":3},{"doubled_letter_count":0,"formula_score":19,"length":12,"letter_phoneme_count_surplus":5,"letters_per_phoneme":"1.71","phoneme_count":7,"rank":9,"row_id":"rank-9","spelling":"breakthrough","tier":1,"vowel_sequence_points":4},{"doubled_letter_count":0,"formula_score":19,"length":13,"letter_phoneme_count_surplus":5,"letters_per_phoneme":"1.63","phoneme_count":8,"rank":10,"row_id":"rank-10","spelling":"breakthroughs","tier":2,"vowel_sequence_points":4},{"doubled_letter_count":2,"formula_score":19,"length":11,"letter_phoneme_count_surplus":5,"letters_per_phoneme":"1.83","phoneme_count":6,"rank":11,"row_id":"rank-11","spelling":"connoisseur","tier":2,"vowel_sequence_points":0},{"doubled_letter_count":0,"formula_score":19,"length":9,"letter_phoneme_count_surplus":5,"letters_per_phoneme":"2.25","phoneme_count":4,"rank":12,"row_id":"rank-12","spelling":"neighbour","tier":1,"vowel_sequence_points":4},{"doubled_letter_count":0,"formula_score":19,"length":10,"letter_phoneme_count_surplus":5,"letters_per_phoneme":"2.00","phoneme_count":5,"rank":13,"row_id":"rank-13","spelling":"neighbours","tier":1,"vowel_sequence_points":4},{"doubled_letter_count":0,"formula_score":19,"length":10,"letter_phoneme_count_surplus":5,"letters_per_phoneme":"2.00","phoneme_count":5,"rank":14,"row_id":"rank-14","spelling":"throughout","tier":1,"vowel_sequence_points":4},{"doubled_letter_count":0,"formula_score":18,"length":11,"letter_phoneme_count_surplus":5,"letters_per_phoneme":"1.83","phoneme_count":6,"rank":15,"row_id":"rank-15","spelling":"neighboring","tier":2,"vowel_sequence_points":3},{"doubled_letter_count":1,"formula_score":18,"length":13,"letter_phoneme_count_surplus":5,"letters_per_phoneme":"1.63","phoneme_count":8,"rank":16,"row_id":"rank-16","spelling":"schoolteacher","tier":2,"vowel_sequence_points":1},{"doubled_letter_count":0,"formula_score":18,"length":11,"letter_phoneme_count_surplus":5,"letters_per_phoneme":"1.83","phoneme_count":6,"rank":17,"row_id":"rank-17","spelling":"slaughtered","tier":1,"vowel_sequence_points":3},{"doubled_letter_count":0,"formula_score":18,"length":12,"letter_phoneme_count_surplus":5,"letters_per_phoneme":"1.71","phoneme_count":7,"rank":18,"row_id":"rank-18","spelling":"slaughtering","tier":2,"vowel_sequence_points":3},{"doubled_letter_count":0,"formula_score":18,"length":8,"letter_phoneme_count_surplus":5,"letters_per_phoneme":"2.67","phoneme_count":3,"rank":19,"row_id":"rank-19","spelling":"thorough","tier":1,"vowel_sequence_points":3},{"doubled_letter_count":0,"formula_score":18,"length":12,"letter_phoneme_count_surplus":5,"letters_per_phoneme":"1.71","phoneme_count":7,"rank":20,"row_id":"rank-20","spelling":"thoroughbred","tier":2,"vowel_sequence_points":3}],"title":"Highest-scoring tier 1-2 exemplars"}],"description":"A reproducible heuristic formula over pronunciation-covered spellings: letter-phoneme count surplus, declared vowel-sequence points, and doubled-letter pairs, with score bands per ranking cohort and a row-limited highest-scoring exemplar.","downloads":[{"contains":["study.spelling-difficulty.population","study.spelling-difficulty.summary","study.spelling-difficulty.bands","study.spelling-difficulty.graphemes","study.spelling-difficulty.highest-scoring-tier-1-2"],"filename":"spelling-difficulty.csv","media_type":"text/csv","sha256":"624261c6cd5f18a14cb3596e4ec24e3ba40f3ac51de51e4b8ff12607bf040122","size_bytes":134899,"status":"available-review-pending"}],"exclusions":[{"criterion_id":"spelling-difficulty.product-policy","description":"Exclude spellings matched by the exact first-party content-safety policy before scoring."},{"criterion_id":"spelling-difficulty.no-pronunciation","description":"Exclude records without a retained pronunciation variant; they receive no score, and nothing is inferred about their spelling difficulty."}],"facts":[{"calculation_id":"research.count-v1","column_id":"record_count","dataset_id":"study.spelling-difficulty.population","denominator":null,"display_value":"59,003","fact_id":"study.spelling-difficulty.analysis-record-count","label":"Analysis population","numerator":null,"row_id":"analysis","unit":"records","value":59003},{"calculation_id":"research.share-v1","column_id":"share_percent","dataset_id":"study.spelling-difficulty.summary","denominator":24183,"display_value":"31.55%","fact_id":"study.spelling-difficulty.tier-1-2-threshold-share","label":"Tier 1-2 covered records scoring at least 6 heuristic points","numerator":7630,"row_id":"tier-1-2-at-least-6","unit":"percent","value":"31.55"},{"calculation_id":"research.maximum-v1","column_id":"score_value","dataset_id":"study.spelling-difficulty.summary","denominator":null,"display_value":"22","fact_id":"study.spelling-difficulty.highest-tier-1-2-score","label":"Highest declared formula score among tier 1-2 covered records","numerator":null,"row_id":"highest-tier-1-2-score","unit":"points","value":22}],"inclusion_criteria":[{"criterion_id":"spelling-difficulty.covered-records","description":"Include safe normalized ASCII spellings with at least one retained CMUdict pronunciation variant, scored against the primary variant."},{"criterion_id":"spelling-difficulty.tier-cohorts","description":"Report the full covered cohort and the tier 1-2 cohort separately; exemplar rows come only from tiers 1-2."}],"initial_cohort":false,"lexical_build_id":"647a998abcbf65af6a16c86cb8460cb4a88761a94544468e9c9e992208fc327c","lexical_manifest_sha256":"46665edef6a77a00080debaa2209f630f4d220ad2888055fa730731f93dc0561","lexical_release_timestamp":"2026-07-26T00:00:00+00:00","limitations":["Every weight is a fixed heuristic choice; the score is not a measured misspelling or reading-error rate.","Only the first retained pronunciation variant is compared; other variants and dialects are ignored.","CMUdict transcribes a primarily North American convention, so letter-phoneme surplus inherits its choices.","The declared vowel-sequence list is finite; spellings outside it can still receive points from letter-phoneme count surplus or doubled letters.","The vowel-sequence and doubled-letter components inspect spelling shape only; they do not compare those sequences with pronunciation.","Tier cohorts are build-specific ranking bands, not universal usage claims."],"narrative":[{"block_id":"study.spelling-difficulty.method","fact_ids":[],"heading":"The scoring formula, stated completely","interpretation_kind":"methodological","paragraphs":["formula score = 3 x max(0, letters - primary phonemes) + declared vowel-grapheme points (longest-first scan) + 2 x doubled-letter pairs. Every weight is a fixed heuristic choice recorded in version control; no component is fitted to reader data. Only the count-surplus component uses pronunciation; the vowel-sequence and doubled-letter components inspect spelling shape."]},{"block_id":"study.spelling-difficulty.result","fact_ids":["study.spelling-difficulty.tier-1-2-threshold-share","study.spelling-difficulty.highest-tier-1-2-score"],"heading":null,"interpretation_kind":"computed","paragraphs":["31.55% of tier 1-2 covered records score at least 6 points under the declared formula; the highest tier 1-2 score is 22."]},{"block_id":"study.spelling-difficulty.scope","fact_ids":[],"heading":"What the score cannot claim","interpretation_kind":"methodological","paragraphs":["The score compares normalized ASCII letters with one retained primary pronunciation variant from a primarily North American source. It is not a measured misspelling rate or validated human difficulty measure, does not resolve dialects or senses, and says nothing about spellings without retained pronunciation evidence. It is not validated for teaching or learning recommendations."]}],"output_license":"No reuse licence asserted; source-rights review pending","output_license_url":"/about/data","population_sha256":"deed50f991280cdb6417c8b525791c86e58fa17db4e73106cf36e0608baa8062","publication_status":"published","related_searches":[{"label":"Browse words by length and pattern","relevance":"Explore the spellings behind the declared formula-score bands in the main word index.","url":"/words"}],"reproducibility_command":"make studies_reproduce STUDY=spelling-difficulty","research_question":"Which pronunciation-covered spellings receive the highest values under the declared three-component formula, and how do its score bands distribute across ranking cohorts?","review":{"approved_analysis_contract_digest":"214ba6800412c85e2bbeb61d6e5c006eb2dbb933ab5082c4f3323745c4ef70d3","approved_content_artifact_digest":"477542693badcafd96a49a013917cb986d413ee48d738561b13d04fa8f330416","approved_content_version":"1.0.1","approved_data_version":"2026.07-lexical.6","approved_lexical_build_id":"647a998abcbf65af6a16c86cb8460cb4a88761a94544468e9c9e992208fc327c","approved_result_sha256":"1b87194a7f3663f2b5ca7ee6b94144f819f8cb6255d4a16bfa295dc50e567151","editorial_approval":"TWI-v1.4.0-editorial-review-2026-08-02-r1","editorial_status":"approved","factual_approval":null,"factual_status":"machine-validated","legal_approval":null,"legal_status":"pending","reevaluate_on":"2026-10-17","reviewed_at":"2026-08-02","reviewer":"Matthew Iles","search_approval":null,"search_eligibility_reason":"Heuristic formula output remains noindex pending source-rights review and separate search approval.","search_eligible":false},"schema_version":1,"slug":"spelling-difficulty","sources":[{"artifact_sha256":"3ed0c94610d8bcf7c11bbb49c56aa49c7234d32b66824df91f554169e572da48","attribution":"dwyl/english-words; the repository carries an Unlicense notice, while its pinned README attributes upstream copyright to Infochimps","establishes":["source-membership","surface-spelling-as-listed"],"legal_review_status":"pending","licence_name":"Unlicense notice; upstream dataset rights unresolved","licence_url":"https://raw.githubusercontent.com/dwyl/english-words/8179fe68775df3f553ef19520db065228e65d1d3/LICENSE.md","limitations":["The pinned repository licence is the Unlicense, but the pinned README says copyright in the extracted source list remains with Infochimps. Human legal review of the upstream rights chain is required.","Do not assert a redistribution right until the Infochimps-to-dwyl rights chain has been reviewed by a human.","Dialect or scope: unspecified"],"redistribution_status":"pending","source_class":"broad-orthographic-list","source_id":"dwyl-english-words","source_version":"git:8179fe68775df3f553ef19520db065228e65d1d3","url":"https://raw.githubusercontent.com/dwyl/english-words/8179fe68775df3f553ef19520db065228e65d1d3/words_alpha.txt"},{"artifact_sha256":"d3fbe8485022088fcf527edcde2fbdc18b4bbc141ac58123c9adb462e086eaf7","attribution":"ENABLE (Enhanced North American Benchmark Lexicon); exact reuse terms are pending legal review","establishes":["source-membership","surface-spelling-as-listed"],"legal_review_status":"pending","licence_name":"Licence terms not stated by the source","licence_url":"https://www.wordgamedictionary.com/enable/","limitations":["The downloaded artifact is content-pinned by SHA-256, but it contains no licence notice and the configured source page does not state reuse terms.","The artifact is reproducibly pinned but no affirmative reuse grant has been located; redistribution approval remains pending.","Dialect or scope: North-American-oriented; exact edition metadata unavailable"],"redistribution_status":"pending","source_class":"historical-game-word-list","source_id":"enable","source_version":"artifact-sha256:d3fbe8485022088fcf527edcde2fbdc18b4bbc141ac58123c9adb462e086eaf7","url":"https://www.wordgamedictionary.com/enable/download/enable.txt"},{"artifact_sha256":"ee2d83651fbb91642bbed2bd30ead404c2cfbdfece01dacf284af6ea47795811","attribution":"first20hours/google-10000-english, derived from the Google Web Trillion Word Corpus and Peter Norvig's compilation","establishes":["source-membership","source-order"],"legal_review_status":"pending","licence_name":"No explicit licence grant in pinned repository metadata","licence_url":"https://raw.githubusercontent.com/first20hours/google-10000-english/bdf4c221bc120b0b7f6c3f1eff1cc1abb975f8d8/LICENSE.md","limitations":["The exact pinned LICENSE.md describes provenance but contains no explicit public-domain dedication or licence grant. Human review is required.","The pinned repository documents provenance but does not supply an affirmative licence grant; do not describe it as public domain.","Dialect or scope: USA no-swears variant"],"redistribution_status":"pending","source_class":"ranked-web-corpus-derived-list","source_id":"google-10000-english","source_version":"git:bdf4c221bc120b0b7f6c3f1eff1cc1abb975f8d8","url":"https://raw.githubusercontent.com/first20hours/google-10000-english/bdf4c221bc120b0b7f6c3f1eff1cc1abb975f8d8/google-10000-english-usa-no-swears.txt"},{"artifact_sha256":"5351ff405b1126ef555791dd4d9798a48e3e9a501a9fc481a9da957752cfb458","attribution":"FrequencyWords content derived from OpenSubtitles, licensed CC BY-SA 4.0","establishes":["source-token-observation","source-token-count"],"legal_review_status":"pending","licence_name":"CC BY-SA 4.0 (content)","licence_url":"https://raw.githubusercontent.com/hermitdave/FrequencyWords/525f9b560de45753a5ea01069454e72e9aa541c6/README.md","limitations":["The exact pinned README states MIT for code and CC BY-SA 4.0 for content. Human review is required for attribution and ShareAlike obligations on compiled and derived data.","Content is declared CC BY-SA 4.0. Attribution and ShareAlike treatment of the compiled frequency evidence and derived tiers require human approval.","Dialect or scope: unspecified"],"redistribution_status":"pending","source_class":"subtitle-derived-token-frequency","source_id":"frequencywords-en-2018","source_version":"git:525f9b560de45753a5ea01069454e72e9aa541c6; corpus-release:2018","url":"https://raw.githubusercontent.com/hermitdave/FrequencyWords/525f9b560de45753a5ea01069454e72e9aa541c6/content/2018/en/en_50k.txt"},{"artifact_sha256":"81917843c7f44ce2b094ac63873c2c7a4cf802040792c455ba3ca406891c3d22","attribution":"CMU Pronouncing Dictionary, Copyright 1993-2015 Carnegie Mellon University, BSD-2-Clause","establishes":["source-membership","arpabet-pronunciation-variant"],"legal_review_status":"pending","licence_name":"BSD-2-Clause","licence_url":"https://raw.githubusercontent.com/cmusphinx/cmudict/74790861f652b15e4ac49015a90074ad62a27690/LICENSE","limitations":["The exact pinned commit includes BSD-2-Clause redistribution terms. Human approval of the production attribution and binary-distribution obligations is still required.","BSD-2-Clause terms are present at the pin; retain copyright and licence notices. Production approval remains a recorded human decision.","Dialect or scope: primarily North American English"],"redistribution_status":"pending","source_class":"pronunciation-dictionary","source_id":"cmudict","source_version":"git:74790861f652b15e4ac49015a90074ad62a27690","url":"https://raw.githubusercontent.com/cmusphinx/cmudict/74790861f652b15e4ac49015a90074ad62a27690/cmudict.dict"},{"artifact_sha256":"ef70bc128a3ae67c23c5a4e31ec6b0ef3780e94881faa3adf9b73c2f7f684b4d","attribution":"The Word Index first-party content-safety exclusion policy","establishes":["content_filter","search_results","result_counts"],"legal_review_status":"pending","licence_name":"Reuse terms not yet approved","licence_url":null,"limitations":["This bounded policy list cannot determine whether every usage is offensive or benign."],"redistribution_status":"pending","source_class":"first-party-content-safety-policy","source_id":"offensive-term-policy","source_version":"2026.07-review-pending.2","url":"/about/data"}],"study_build_id":"cbbdba7067eaa6317e060c5795ae8c7db9b641eea209d5ef5825773634508b5e","study_id":"study.spelling-difficulty","study_version":"1.0.1","tables":[{"caption":"The exact population used by this version of the analysis.","column_ids":["stage","record_count"],"dataset_id":"study.spelling-difficulty.population","heading":"Population and exclusions","note":null,"row_ids":["raw","policy-excluded","safe","analysis"],"table_id":"study.spelling-difficulty.population-table"},{"caption":"Every value derives from the declared integer scoring formula.","column_ids":["metric","record_count","denominator","share_percent","score_value"],"dataset_id":"study.spelling-difficulty.summary","heading":"Headline formula-score results","note":null,"row_ids":["covered-tier-1-2","tier-1-2-at-least-6","highest-tier-1-2-score","highest-all-score"],"table_id":"study.spelling-difficulty.summary-table"},{"caption":"Bands partition the integer score; tier cohorts are build-specific ranking bands, not usage claims.","column_ids":["band","score_range","all_record_count","all_share_percent","common_record_count","common_share_percent"],"dataset_id":"study.spelling-difficulty.bands","heading":"Formula-score bands by ranking cohort","note":null,"row_ids":["band-0","band-1-2","band-3-5","band-6-9","band-10-plus"],"table_id":"study.spelling-difficulty.bands-table"},{"caption":"The scan is deterministic: longest declared grapheme wins per position.","column_ids":["grapheme","weight","record_count","share_percent"],"dataset_id":"study.spelling-difficulty.graphemes","heading":"Declared vowel graphemes and their prevalence","note":null,"row_ids":["grapheme-ae","grapheme-aigh","grapheme-augh","grapheme-ea","grapheme-eau","grapheme-ei","grapheme-eigh","grapheme-eou","grapheme-ie","grapheme-ieu","grapheme-igh","grapheme-iou","grapheme-oe","grapheme-ou","grapheme-ough","grapheme-ue","grapheme-ui"],"table_id":"study.spelling-difficulty.grapheme-table"},{"caption":"A computed exemplar limited to 20 rows; ties break alphabetically.","column_ids":["rank","spelling","length","phoneme_count","letters_per_phoneme","letter_phoneme_count_surplus","vowel_sequence_points","doubled_letter_count","formula_score"],"dataset_id":"study.spelling-difficulty.highest-scoring-tier-1-2","heading":"Highest-scoring tier 1-2 spellings","note":null,"row_ids":["rank-1","rank-2","rank-3","rank-4","rank-5","rank-6","rank-7","rank-8","rank-9","rank-10","rank-11","rank-12","rank-13","rank-14","rank-15","rank-16","rank-17","rank-18","rank-19","rank-20"],"table_id":"study.spelling-difficulty.exemplar-table"}],"title":"Declared spelling-form score: a reproducible three-component formula","transformation_steps":[{"description":"Compare normalized letter counts with the primary retained variant's phoneme count; each surplus letter contributes three declared points.","deterministic":true,"limitations":["A count surplus cannot identify silent letters or align individual letters to phonemes."],"transformation_id":"spelling-difficulty.surplus-v1"},{"description":"Scan each spelling left to right against the declared vowel-sequence list, longest sequence first, summing declared weights.","deterministic":true,"limitations":["The scan counts spelling shapes only; it does not verify each grapheme's pronunciation."],"transformation_id":"spelling-difficulty.grapheme-scan-v1"},{"description":"Count adjacent identical letter pairs; each pair contributes two declared points.","deterministic":true,"limitations":[],"transformation_id":"spelling-difficulty.doubling-v1"},{"description":"Partition total formula scores into five declared numeric bands and rank the tier 1-2 exemplar by score with alphabetical tie-breaks, limited to 20 rows.","deterministic":true,"limitations":["Band boundaries are declared editorial cut points on an integer scale."],"transformation_id":"spelling-difficulty.banding-v1"}],"update_history":[{"data_version":"2026.07-lexical.4","effective_date":"2026-07-19","result_sha256":"adc8445a55a61610b21469db2ed536e3cefe2c9e88918eaf0393117ac084bc1c","review_status":"machine-validated","reviewer":null,"study_version":"1.0.0","summary":"Initial reproducible heuristic difficulty release."},{"data_version":"2026.07-lexical.5","effective_date":"2026-07-25","result_sha256":"adc8445a55a61610b21469db2ed536e3cefe2c9e88918eaf0393117ac084bc1c","review_status":"machine-validated","reviewer":null,"study_version":"1.0.0","summary":"Rebound byte-identical results to the 2026.07-lexical.5 data release; no numeric changes."},{"data_version":"2026.07-lexical.6","effective_date":"2026-07-26","result_sha256":"adc8445a55a61610b21469db2ed536e3cefe2c9e88918eaf0393117ac084bc1c","review_status":"machine-validated","reviewer":null,"study_version":"1.0.0","summary":"Rebound byte-identical results to the 2026.07-lexical.6 data release; no numeric changes."},{"data_version":"2026.07-lexical.6","effective_date":"2026-08-01","result_sha256":"1b87194a7f3663f2b5ca7ee6b94144f819f8cb6255d4a16bfa295dc50e567151","review_status":"machine-validated","reviewer":null,"study_version":"1.0.1","summary":"Reframed the output as an unvalidated declared formula score, renamed its fields and bands, and limited instructional claims."}],"user_relevance":"Shows exactly how the declared formula ranks tier 1-2 spellings in this build; the score is an audit and exploration measure, not validated spelling difficulty or teaching guidance."}
