{
 "_generated": "scripts/build-exports.mjs",
 "_source": "Dukes, Kais. The Quranic Arabic Corpus, version 0.4. Leeds: Language Research Group, University of Leeds, 2009-2017. https://corpus.quran.com/. GNU GPL.",
 "_chronologySource": "Egyptian Standard (Cairo 1924) revelation order, four-period classification following the Nöldeke-Bell tradition (Watt, \"Bell's Introduction to the Qur'an\", 1970).",
 "_computed": "2026-08-09",
 "tables": {
  "root-frequencies": {
   "description": "Every one of the 1,642 roots: raw occurrence count, overall normalized frequency, and per-period count and normalized frequency.",
   "rowCount": 1642,
   "countingRule": "Root occurrence = token count from Leeds morphology (root field non-empty). Normalized frequency = (count / total tokens for the given scope) * 1000, i.e. occurrences per 1,000 tokens.",
   "verification": "totalCount and normalizedFrequencyOverall: Verified (direct computation). Per-period fields: Nuanced (depend on the four-period chronology named in _chronologySource).",
   "fields": [
    {
     "name": "root",
     "type": "string",
     "unit": null,
     "description": "Buckwalter-transliterated root."
    },
    {
     "name": "safeKey",
     "type": "string",
     "unit": null,
     "description": "URL/filename-safe encoding of root, used to link to data/association/{safeKey}.json and roots.html?root={safeKey}."
    },
    {
     "name": "arabic",
     "type": "string",
     "unit": null,
     "description": "Root letters in Arabic script, space-separated."
    },
    {
     "name": "rootLatin",
     "type": "string",
     "unit": null,
     "description": "Root in Latin transliteration with diacritics."
    },
    {
     "name": "totalCount",
     "type": "integer",
     "unit": "tokens",
     "description": "Total occurrences of this root across the whole corpus (77,429 tokens)."
    },
    {
     "name": "normalizedFrequencyOverall",
     "type": "number",
     "unit": "occurrences per 1,000 tokens",
     "description": "(totalCount / 77,429) * 1000."
    },
    {
     "name": "count_meccan-early",
     "type": "integer",
     "unit": "tokens",
     "description": "Occurrences of this root in the Early Meccan period."
    },
    {
     "name": "normalizedFrequency_meccan-early",
     "type": "number",
     "unit": "occurrences per 1,000 tokens",
     "description": "(count_meccan-early / period token total) * 1000."
    },
    {
     "name": "count_meccan-middle",
     "type": "integer",
     "unit": "tokens",
     "description": "Occurrences of this root in the Middle Meccan period."
    },
    {
     "name": "normalizedFrequency_meccan-middle",
     "type": "number",
     "unit": "occurrences per 1,000 tokens",
     "description": "(count_meccan-middle / period token total) * 1000."
    },
    {
     "name": "count_meccan-late",
     "type": "integer",
     "unit": "tokens",
     "description": "Occurrences of this root in the Late Meccan period."
    },
    {
     "name": "normalizedFrequency_meccan-late",
     "type": "number",
     "unit": "occurrences per 1,000 tokens",
     "description": "(count_meccan-late / period token total) * 1000."
    },
    {
     "name": "count_medinan",
     "type": "integer",
     "unit": "tokens",
     "description": "Occurrences of this root in the Medinan period."
    },
    {
     "name": "normalizedFrequency_medinan",
     "type": "number",
     "unit": "occurrences per 1,000 tokens",
     "description": "(count_medinan / period token total) * 1000."
    }
   ]
  },
  "association-pairs": {
   "description": "Root-pair association statistics: the union of every pair appearing in any root's top-25-by-LLR partner list, deduplicated by unordered pair.",
   "rowCount": 5211,
   "countingRule": "Co-occurrence counted at the verse level over N = 6,236 verses. Only pairs with at least 5 shared verses were computed by scripts/compute-association-stats.mjs; only the top 25 partners per root (by LLR) are represented here, so this is not the complete set of all pairs meeting the 5-shared-verse threshold.",
   "verification": "Verified: direct computation from Leeds morphology, cross-checked against a hand-computed 2x2 table and against data/cooccurrence/*.json's independently computed counts.",
   "fields": [
    {
     "name": "rootA",
     "type": "string",
     "unit": null,
     "description": "First root of the pair (Buckwalter)."
    },
    {
     "name": "rootASafeKey",
     "type": "string",
     "unit": null,
     "description": "URL/filename-safe encoding of rootA."
    },
    {
     "name": "rootALatin",
     "type": "string",
     "unit": null,
     "description": "rootA in Latin transliteration."
    },
    {
     "name": "rootB",
     "type": "string",
     "unit": null,
     "description": "Second root of the pair (Buckwalter)."
    },
    {
     "name": "rootBSafeKey",
     "type": "string",
     "unit": null,
     "description": "URL/filename-safe encoding of rootB."
    },
    {
     "name": "rootBLatin",
     "type": "string",
     "unit": null,
     "description": "rootB in Latin transliteration."
    },
    {
     "name": "sharedVerses",
     "type": "integer",
     "unit": "verses",
     "description": "k11: number of verses in which both roots are attested."
    },
    {
     "name": "pmi",
     "type": "number",
     "unit": "bits (log base 2)",
     "description": "Pointwise mutual information: log2((k11*N)/((k11+k12)*(k11+k21))), N=6,236, rounded to 2 decimals."
    },
    {
     "name": "dice",
     "type": "number",
     "unit": null,
     "description": "Dice coefficient: 2*k11/(2*k11+k12+k21), rounded to 3 decimals."
    },
    {
     "name": "llr",
     "type": "number",
     "unit": null,
     "description": "Dunning's log-likelihood ratio (G2) over the pair's verse-level 2x2 table, rounded to 2 decimals."
    }
   ]
  },
  "surah-stats": {
   "description": "Per-surah corpus fingerprint: all 114 surahs.",
   "rowCount": 114,
   "countingRule": "verseCount/tokenCount/distinctRootCount and diversity ratios are read from data/surah-profiles.json (Leeds morphology tally per surah). revelationOrder and period are read from data/chronology.json. formMATTR/formMTLD are length-robust alternatives to the raw formDiversityRatio type-token ratio (scripts/lib/lexical-diversity.mjs), included because raw TTR is mechanically confounded by surah length.",
   "verification": "verseCount/tokenCount/distinctRootCount/diversity ratios/nounPct/verbPct: Verified (direct computation). formMATTR/formMTLD: Verified (direct computation; formulas hand-verified against fixed-point fixtures). revelationOrder/period: Nuanced (Cairo 1924 / Nöldeke-Bell chronology, one scheme among several).",
   "fields": [
    {
     "name": "surah",
     "type": "integer",
     "unit": null,
     "description": "Surah number, 1-114, Cairo (mushaf) order."
    },
    {
     "name": "nameTranslit",
     "type": "string",
     "unit": null,
     "description": "Transliterated surah name."
    },
    {
     "name": "nameArabic",
     "type": "string",
     "unit": null,
     "description": "Surah name in Arabic script."
    },
    {
     "name": "nameEnglish",
     "type": "string",
     "unit": null,
     "description": "English meaning of the surah name."
    },
    {
     "name": "revelationOrder",
     "type": "integer",
     "unit": null,
     "description": "Position in the Cairo 1924 revelation-order sequence (1 = first revealed)."
    },
    {
     "name": "period",
     "type": "string",
     "unit": null,
     "description": "One of meccan-early, meccan-middle, meccan-late, medinan (Nöldeke-Bell four-period classification)."
    },
    {
     "name": "verseCount",
     "type": "integer",
     "unit": "verses",
     "description": "Number of verses (ayat) in the surah, Cairo numbering."
    },
    {
     "name": "tokenCount",
     "type": "integer",
     "unit": "tokens",
     "description": "Total Leeds morphological tokens in the surah."
    },
    {
     "name": "distinctRootCount",
     "type": "integer",
     "unit": "roots",
     "description": "Count of distinct roots attested in the surah."
    },
    {
     "name": "rootDiversityRatio",
     "type": "number",
     "unit": null,
     "description": "distinctRootCount / tokenCount."
    },
    {
     "name": "distinctFormCount",
     "type": "integer",
     "unit": "forms",
     "description": "Count of distinct surface (written) forms in the surah."
    },
    {
     "name": "formDiversityRatio",
     "type": "number",
     "unit": null,
     "description": "distinctFormCount / tokenCount. Mechanically declines as tokenCount grows (a sample-size artifact); see formMATTR/formMTLD for length-robust alternatives."
    },
    {
     "name": "formMATTR",
     "type": "number",
     "unit": null,
     "description": "Moving-average type-token ratio (Covington & McFall 2010) over the surah's ordered surface-form tokens, 25-token window. Null for the 9 surahs shorter than the window."
    },
    {
     "name": "formMTLD",
     "type": "number",
     "unit": "tokens",
     "description": "Measure of Textual Lexical Diversity (McCarthy & Jarvis 2010): mean tokens-per-factor at a 0.72 TTR threshold, bidirectionally averaged. Null only if the surah's running TTR never reaches the threshold."
    },
    {
     "name": "distinctLemmaCount",
     "type": "integer",
     "unit": "lemmas",
     "description": "Count of distinct lemmas in the surah."
    },
    {
     "name": "lemmaDiversityRatio",
     "type": "number",
     "unit": null,
     "description": "distinctLemmaCount / tokenCount."
    },
    {
     "name": "nounPct",
     "type": "number",
     "unit": "percent",
     "description": "Percentage of the surah's tokens tagged noun (N) by Leeds POS tagging."
    },
    {
     "name": "verbPct",
     "type": "number",
     "unit": "percent",
     "description": "Percentage of the surah's tokens tagged verb (V) by Leeds POS tagging."
    }
   ]
  },
  "verse-lengths": {
   "description": "Every verse in the corpus (6,236 rows) with its token length and revelation period.",
   "rowCount": 6236,
   "countingRule": "tokens = number of Leeds morphological tokens in the verse (data/morphology/{surah}.json entry length). Ordered by surah, then verse.",
   "verification": "surah/verse/tokens: Verified (direct tally). period: Nuanced (Cairo 1924 / Nöldeke-Bell chronology).",
   "fields": [
    {
     "name": "surah",
     "type": "integer",
     "unit": null,
     "description": "Surah number, 1-114."
    },
    {
     "name": "verse",
     "type": "integer",
     "unit": null,
     "description": "Verse (ayah) number within the surah, Cairo numbering."
    },
    {
     "name": "tokens",
     "type": "integer",
     "unit": "tokens",
     "description": "Number of Leeds morphological tokens in this verse."
    },
    {
     "name": "period",
     "type": "string",
     "unit": null,
     "description": "One of meccan-early, meccan-middle, meccan-late, medinan; null if the surah has no chronology entry."
    }
   ]
  },
  "formulas": {
   "description": "Every recurring 3-5 word sequence in the Qur'an (18,408 rows: 6,403 root-view + 12,005 surface-view), with its first occurrence.",
   "rowCount": 18408,
   "countingRule": "A sequence recurs if it appears 2+ times, counted two independent ways: root stream (words reduced to their consonantal root; unrooted particles skipped, so matched words need not be consecutive) and surface stream (diacritic-stripped written form, all words included, always consecutive). Only the first occurrence's location is included here; full occurrence lists are in data/formulas-root.json and data/formulas-surface.json.",
   "verification": "Verified (direct computation from Leeds morphology).",
   "fields": [
    {
     "name": "stream",
     "type": "string",
     "unit": null,
     "description": "root or surface."
    },
    {
     "name": "n",
     "type": "integer",
     "unit": "words",
     "description": "Sequence length, 3-5."
    },
    {
     "name": "display",
     "type": "string",
     "unit": null,
     "description": "Root stream: dot-separated Latin transliteration. Surface stream: the Arabic phrase itself."
    },
    {
     "name": "arabic",
     "type": "string",
     "unit": null,
     "description": "Arabic script for the sequence (root stream: root letters; surface stream: same as display)."
    },
    {
     "name": "count",
     "type": "integer",
     "unit": "occurrences",
     "description": "Total occurrences of this sequence across the corpus."
    },
    {
     "name": "firstSurah",
     "type": "integer",
     "unit": null,
     "description": "Surah of the sequence's first occurrence."
    },
    {
     "name": "firstVerse",
     "type": "integer",
     "unit": null,
     "description": "Verse of the sequence's first occurrence."
    }
   ]
  },
  "centrality": {
   "description": "Network centrality for all 1,642 roots over the root co-occurrence graph (5,211 edges, built from each root's top-25-by-LLR partners).",
   "rowCount": 1642,
   "countingRule": "Degree, weighted degree (sum of incident LLR weights), betweenness (Brandes' algorithm, unweighted shortest paths), and eigenvector centrality (power iteration on the LLR-weighted adjacency matrix), each with its rank among all 1,642 roots. Method detail in data/centrality/methods.json.",
   "verification": "Nuanced: the four measures rank roots differently by design, and the graph itself is a subset (each root's top-25 partners, not every pair meeting the underlying 5-shared-verse threshold).",
   "fields": [
    {
     "name": "root",
     "type": "string",
     "unit": null,
     "description": "Buckwalter-transliterated root."
    },
    {
     "name": "safeKey",
     "type": "string",
     "unit": null,
     "description": "URL/filename-safe encoding of root."
    },
    {
     "name": "rootLatin",
     "type": "string",
     "unit": null,
     "description": "Root in Latin transliteration with diacritics."
    },
    {
     "name": "degree",
     "type": "integer",
     "unit": "neighbors",
     "description": "Count of distinct partner roots."
    },
    {
     "name": "degreeRank",
     "type": "integer",
     "unit": null,
     "description": "Rank by degree, 1 = highest, among 1,642 roots."
    },
    {
     "name": "weightedDegree",
     "type": "number",
     "unit": null,
     "description": "Sum of incident edge weights (LLR)."
    },
    {
     "name": "weightedDegreeRank",
     "type": "integer",
     "unit": null,
     "description": "Rank by weighted degree."
    },
    {
     "name": "betweenness",
     "type": "number",
     "unit": null,
     "description": "Betweenness centrality (unweighted shortest paths)."
    },
    {
     "name": "betweennessRank",
     "type": "integer",
     "unit": null,
     "description": "Rank by betweenness."
    },
    {
     "name": "eigenvector",
     "type": "number",
     "unit": null,
     "description": "Eigenvector centrality (LLR-weighted, power iteration, L2-normalized)."
    },
    {
     "name": "eigenvectorRank",
     "type": "integer",
     "unit": null,
     "description": "Rank by eigenvector centrality."
    }
   ]
  },
  "rhyme-summary": {
   "description": "Per-surah roll-up of verse-ending (rhyme) patterns for all 114 surahs.",
   "rowCount": 114,
   "countingRule": "Verse-final word per verse, diacritics/tatweel stripped (an orthographic proxy for pausal form, not a phonological transcription). familyCount/dominantKey/dominantShare/shiftCount are over the fine rhyme key (last two letters after collapsing hamza seats). meanRunLength = verseCount / (shiftCount + 1). Full method note and per-verse detail in data/rhyme/{surah}.json.",
   "verification": "Verified (direct computation from Leeds morphology and the Tanzil Uthmani text).",
   "fields": [
    {
     "name": "surah",
     "type": "integer",
     "unit": null,
     "description": "Surah number, 1-114."
    },
    {
     "name": "verseCount",
     "type": "integer",
     "unit": "verses",
     "description": "Number of verses in the surah."
    },
    {
     "name": "familyCount",
     "type": "integer",
     "unit": null,
     "description": "Count of distinct fine-key rhyme families in the surah."
    },
    {
     "name": "dominantKey",
     "type": "string",
     "unit": null,
     "description": "The most frequent fine rhyme key in the surah."
    },
    {
     "name": "dominantShare",
     "type": "number",
     "unit": null,
     "description": "Share of verses ending on the dominant key, 0-1."
    },
    {
     "name": "shiftCount",
     "type": "integer",
     "unit": null,
     "description": "Number of verse-to-verse changes in the fine rhyme key."
    },
    {
     "name": "topRefrainPausal",
     "type": "string",
     "unit": null,
     "description": "Pausal form of the most-repeated verse ending recurring 3+ times, if any; null otherwise."
    },
    {
     "name": "topRefrainCount",
     "type": "integer",
     "unit": null,
     "description": "Occurrences of topRefrainPausal; null if there is no refrain."
    },
    {
     "name": "meanRunLength",
     "type": "number",
     "unit": "verses",
     "description": "verseCount / (shiftCount + 1): average consecutive-verse run on one ending before it changes."
    }
   ]
  },
  "fawatih": {
   "description": "The 29 surahs opening with a sequence of isolated letters (fawatih / al-muqatta'at), and which combination.",
   "rowCount": 29,
   "countingRule": "Detected from the Leeds morphology: a surah's opening verse consisting solely of isolated-letter tokens. 14 distinct letter combinations recur across the 29 surahs.",
   "verification": "Verified (direct detection from Leeds morphology). The letters' meaning is not asserted; the classical tradition has never settled it.",
   "fields": [
    {
     "name": "surah",
     "type": "integer",
     "unit": null,
     "description": "Surah number."
    },
    {
     "name": "verse",
     "type": "integer",
     "unit": null,
     "description": "Verse carrying the isolated letters (always 1)."
    },
    {
     "name": "letters",
     "type": "string",
     "unit": null,
     "description": "The isolated letters in Arabic script, as written."
    }
   ]
  },
  "discursive-pivots": {
   "description": "137 verses mechanically flagged for opening with a temporal particle (idh or idha) while sharing a content root with the immediately preceding verse.",
   "rowCount": 137,
   "countingRule": "A verse qualifies if its first content word is idh or idha (Leeds lemma) and it shares at least one non-stoplisted root with the previous verse. A candidate marker of a discursive turn, not a scholar's identification of one; method note in data/discursive-pivots.json.",
   "verification": "Nuanced: depends on the fixed marker list and content-root stoplist; a different choice of either would change the count.",
   "fields": [
    {
     "name": "surah",
     "type": "integer",
     "unit": null,
     "description": "Surah number."
    },
    {
     "name": "verse",
     "type": "integer",
     "unit": null,
     "description": "The flagged verse."
    },
    {
     "name": "marker",
     "type": "string",
     "unit": null,
     "description": "The temporal particle opening the verse: idh or idha."
    },
    {
     "name": "previousVerse",
     "type": "integer",
     "unit": null,
     "description": "The preceding verse the flagged verse shares a root with."
    },
    {
     "name": "sharedRoots",
     "type": "string",
     "unit": null,
     "description": "Semicolon-separated list of the shared root(s) in Latin transliteration."
    }
   ]
  },
  "structure": {
   "description": "Mechanically segmented sections for all 114 surahs (TextTiling-derived changepoint detection over lexical cohesion), not a transcribed scholarly outline.",
   "rowCount": 417,
   "countingRule": "Per-verse boundary scores from five weighted signals (rhyme-family change, verse-length discontinuity, lexical-cohesion drop, formula onset, discursive-pivot markers); section count set by a per-surah significance threshold, not a fixed target. 34 of 114 surahs get exactly one section (no boundary cleared the threshold). Full method and per-boundary evidence in data/structure/{surah}.json.",
   "verification": "Nuanced: a computed segmentation, not a scholar's reading; never attributed to any named scholar. See docs/maintainer-guide.md on the named-scholar outline policy.",
   "fields": [
    {
     "name": "surah",
     "type": "integer",
     "unit": null,
     "description": "Surah number."
    },
    {
     "name": "sectionIndex",
     "type": "integer",
     "unit": null,
     "description": "1-based section number within the surah."
    },
    {
     "name": "fromVerse",
     "type": "integer",
     "unit": null,
     "description": "First verse of the section."
    },
    {
     "name": "toVerse",
     "type": "integer",
     "unit": null,
     "description": "Last verse of the section."
    },
    {
     "name": "verseCount",
     "type": "integer",
     "unit": "verses",
     "description": "Number of verses in the section."
    }
   ]
  },
  "structure-tests": {
   "description": "Four block-level mirror-symmetry tests (concentric pairing, inclusio, formula bookending, verse-length symmetry) over the computed sections in the structure table, one row per surah.",
   "rowCount": 114,
   "countingRule": "Each test's p-value comes from a block-order permutation null (blocks kept intact, 10,000 seeded shuffles, or exact enumeration when feasible), corrected jointly across all 345 candidates from all four tests via Benjamini-Hochberg (q<0.05). Null for surahs with too few sections for a given test. 0 of 345 candidates reached significance after correction. Full method in data/structure-tests.json.",
   "verification": "Nuanced: a null result describes this specific mechanical test over a computed segmentation, not the scholarly literature on ring composition, which this site never asserts an outline from.",
   "fields": [
    {
     "name": "surah",
     "type": "integer",
     "unit": null,
     "description": "Surah number."
    },
    {
     "name": "verseCount",
     "type": "integer",
     "unit": "verses",
     "description": "Verses in the surah."
    },
    {
     "name": "sections",
     "type": "integer",
     "unit": null,
     "description": "Number of computed sections (from the structure table)."
    },
    {
     "name": "concentricParallelism_observed",
     "type": "number",
     "unit": null,
     "description": "Observed mean Jaccard similarity of mirrored section pairs; null if the surah has too few sections for this test."
    },
    {
     "name": "concentricParallelism_pValue",
     "type": "number",
     "unit": null,
     "description": "Permutation p-value for concentricParallelism; null if not applicable."
    },
    {
     "name": "concentricParallelism_survivor",
     "type": "boolean",
     "unit": null,
     "description": "Whether this candidate survived the pooled Benjamini-Hochberg correction; null if not applicable."
    },
    {
     "name": "inclusio_observed",
     "type": "number",
     "unit": null,
     "description": "Observed vocabulary overlap between the first and last section; null if not applicable."
    },
    {
     "name": "inclusio_pValue",
     "type": "number",
     "unit": null,
     "description": "Permutation p-value for inclusio; null if not applicable."
    },
    {
     "name": "inclusio_survivor",
     "type": "boolean",
     "unit": null,
     "description": "Whether this candidate survived correction; null if not applicable."
    },
    {
     "name": "formulaBookending_observed",
     "type": "number",
     "unit": null,
     "description": "Observed formula-bracketing statistic; null if not applicable."
    },
    {
     "name": "formulaBookending_pValue",
     "type": "number",
     "unit": null,
     "description": "Permutation p-value for formulaBookending; null if not applicable."
    },
    {
     "name": "formulaBookending_survivor",
     "type": "boolean",
     "unit": null,
     "description": "Whether this candidate survived correction; null if not applicable."
    },
    {
     "name": "lengthSymmetry_observed",
     "type": "number",
     "unit": null,
     "description": "Observed correlation of the verse-length profile with its reverse; null if not applicable."
    },
    {
     "name": "lengthSymmetry_pValue",
     "type": "number",
     "unit": null,
     "description": "Permutation p-value for lengthSymmetry; null if not applicable."
    },
    {
     "name": "lengthSymmetry_survivor",
     "type": "boolean",
     "unit": null,
     "description": "Whether this candidate survived correction; null if not applicable."
    }
   ]
  },
  "theme-surah-density": {
   "description": "Sparse theme-by-surah matrix: for each surah, the themes whose root-family vocabulary clusters most densely in it.",
   "rowCount": 252,
   "countingRule": "perThousand = theme-root tokens per 1,000 surah tokens (Leeds counts, minimum 2 tokens). Each theme lists at most its top 8 surahs by density, so a surah's absence from this table for a given theme means it is not among that theme's densest, not that the vocabulary is absent entirely. Root-to-theme grouping is editorial (see themes.html); the counting is mechanical.",
   "verification": "Nuanced: perThousand is a direct computation, but which roots belong to which theme is an editorial classification, not a computed fact.",
   "fields": [
    {
     "name": "surah",
     "type": "integer",
     "unit": null,
     "description": "Surah number."
    },
    {
     "name": "themeSlug",
     "type": "string",
     "unit": null,
     "description": "Theme identifier, matches themes.html's slug."
    },
    {
     "name": "themeTitle",
     "type": "string",
     "unit": null,
     "description": "Theme display title."
    },
    {
     "name": "perThousand",
     "type": "number",
     "unit": "theme-root tokens per 1,000 surah tokens",
     "description": "Density of this theme's root family in this surah."
    }
   ]
  },
  "formulaic-density": {
   "description": "Per-surah mean share of words covered by a recurring 3-5-word phrase (Bannister's oral-formulaic density), tested against a length-matched null, for all 114 surahs.",
   "rowCount": 114,
   "countingRule": "meanDensity{Root,Surface} = unweighted mean of per-verse density (covered word positions / verse token count) across the surah. p-value: one-sided, 10,000 draws resampling verseCount verses with replacement from the pooled corpus per-verse densities. survivor: true if this candidate is among the 228 (114 surahs x 2 streams) jointly Benjamini-Hochberg corrected at q<0.05. Per-verse detail and method in data/formulaic-density.json.",
   "verification": "Nuanced: the significance test's null (uniform resampling across the whole corpus) does not control for genre or period, only length.",
   "fields": [
    {
     "name": "surah",
     "type": "integer",
     "unit": null,
     "description": "Surah number."
    },
    {
     "name": "verseCount",
     "type": "integer",
     "unit": null,
     "description": "Verses in this surah."
    },
    {
     "name": "meanDensityRoot",
     "type": "number",
     "unit": null,
     "description": "Mean per-verse root-stream formulaic density, 0-1."
    },
    {
     "name": "pValueRoot",
     "type": "number",
     "unit": null,
     "description": "One-sided permutation p-value, root stream."
    },
    {
     "name": "survivorRoot",
     "type": "boolean",
     "unit": null,
     "description": "Survives the pooled BH-FDR correction at q<0.05, root stream."
    },
    {
     "name": "meanDensitySurface",
     "type": "number",
     "unit": null,
     "description": "Mean per-verse surface-stream formulaic density, 0-1."
    },
    {
     "name": "pValueSurface",
     "type": "number",
     "unit": null,
     "description": "One-sided permutation p-value, surface stream."
    },
    {
     "name": "survivorSurface",
     "type": "boolean",
     "unit": null,
     "description": "Survives the pooled BH-FDR correction at q<0.05, surface stream."
    }
   ]
  },
  "dispersion": {
   "description": "How evenly each of the 1,642 roots is spread across the 114 surahs, weighted by surah token count.",
   "rowCount": 1642,
   "countingRule": "dp/dpNorm: Gries's Deviation of Proportions (Gries 2008) and its corrected normalization (Lijffijt & Gries 2012). juillandD: Juilland & Chang-Rodriguez (1964), over each surah's per-1,000-token rate; classically assumes comparably-sized parts, which this corpus's 114 surahs are not. adjustedFrequency = totalCount * (1 - dp). Method detail in data/dispersion/methods.json.",
   "verification": "Nuanced: dp/dpNorm and juillandD are independent formulas that can and do disagree; this site reports both rather than picking one as authoritative.",
   "fields": [
    {
     "name": "root",
     "type": "string",
     "unit": null,
     "description": "Buckwalter-transliterated root."
    },
    {
     "name": "safeKey",
     "type": "string",
     "unit": null,
     "description": "URL/filename-safe encoding of root."
    },
    {
     "name": "rootLatin",
     "type": "string",
     "unit": null,
     "description": "Root in Latin transliteration with diacritics."
    },
    {
     "name": "totalCount",
     "type": "integer",
     "unit": "occurrences",
     "description": "Corpus-wide occurrence count."
    },
    {
     "name": "surahsOccurringIn",
     "type": "integer",
     "unit": null,
     "description": "Count of the 114 surahs the root occurs in at least once."
    },
    {
     "name": "dp",
     "type": "number",
     "unit": null,
     "description": "Gries's Deviation of Proportions. Range [0, 1 - min part share]. Attains 1 - min part share exactly when all occurrences fall in the corpus's smallest part."
    },
    {
     "name": "dpNorm",
     "type": "number",
     "unit": null,
     "description": "DP rescaled to [0, 1] via the Lijffijt & Gries (2012) correction. Attains 1 at the same extreme as dp's own maximum."
    },
    {
     "name": "juillandD",
     "type": "number",
     "unit": null,
     "description": "Juilland's D; 1 = perfectly even. Not clamped, can be negative."
    },
    {
     "name": "adjustedFrequency",
     "type": "number",
     "unit": null,
     "description": "totalCount * (1 - dp): raw frequency discounted for clumping."
    }
   ]
  }
 }
}
