<?xml version="1.0" encoding="UTF-8"?>
<resource xmlns="http://datacite.org/schema/kernel-4" xmlns:xsi="http://www.w3.org/2001/XMLSchema-instance" xsi:schemaLocation="http://datacite.org/schema/kernel-4 http://schema.datacite.org/meta/kernel-4.5/metadata.xsd">
  <identifier identifierType="DOI">10.18710/FVHTFM</identifier>
  <creators>
    <creator>
      <creatorName nameType="Personal">Sönning, Lukas</creatorName>
      <givenName>Lukas</givenName>
      <familyName>Sönning</familyName>
      <nameIdentifier nameIdentifierScheme="ORCID" schemeURI="https://orcid.org">https://orcid.org/0000-0002-2705-395X</nameIdentifier>
      <affiliation>University of Bamberg</affiliation>
    </creator>
  </creators>
  <titles>
    <title>Background data for: Advancing our understanding of dispersion measures in corpus research</title>
  </titles>
  <publisher>DataverseNO</publisher>
  <publicationYear>2024</publicationYear>
  <subjects>
    <subject>Arts and Humanities</subject>
    <subject>dispersion</subject>
    <subject>corpus linguistics</subject>
    <subject>methodology</subject>
    <subject>corpus design</subject>
    <subject>Brown Corpus</subject>
    <subject>dispersion measures</subject>
    <subject>lexical dispersion</subject>
    <subject>word importance</subject>
    <subject>vocabulary lists</subject>
    <subject>word frequency lists</subject>
    <subject>text-level analysis</subject>
    <subject>frequency</subject>
    <subject>Juilland&amp;apos;s D</subject>
    <subject>Gries&amp;apos; DP</subject>
    <subject>DA</subject>
    <subject>English</subject>
  </subjects>
  <contributors>
    <contributor contributorType="Producer">
      <contributorName nameType="Organizational">University of Bamberg</contributorName>
    </contributor>
    <contributor contributorType="Distributor">
      <contributorName nameType="Personal">The Tromsø Repository of Language and Linguistics (TROLLing)</contributorName>
      <givenName>The</givenName>
      <familyName>Tromsø Repository of Language and Linguistics (TROLLing)</familyName>
    </contributor>
    <contributor contributorType="ContactPerson">
      <contributorName nameType="Personal">Sönning, Lukas</contributorName>
      <givenName>Lukas</givenName>
      <familyName>Sönning</familyName>
      <affiliation>University of Bamberg</affiliation>
    </contributor>
  </contributors>
  <dates>
    <date dateType="Created">2023-06-28</date>
    <date dateType="Submitted">2023-12-19</date>
    <date dateType="Available">2024-11-26</date>
    <date dateType="Updated">2025-07-17</date>
    <date dateType="Collected">2023-06-14/2023-06-28</date>
    <date dateType="Other" dateInformation="Time period covered by the data">1961-01-01/1961-12-31</date>
  </dates>
  <resourceType resourceTypeGeneral="Dataset">textual linguistic data;corpus data;observational data</resourceType>
  <relatedIdentifiers>
    <relatedIdentifier relationType="IsSupplementTo" relatedIdentifierType="DOI">10.3366/COR.2025.0326</relatedIdentifier>
    <relatedIdentifier relationType="HasPart" relatedIdentifierType="DOI">10.18710/FVHTFM/BHA3YS</relatedIdentifier>
    <relatedIdentifier relationType="HasPart" relatedIdentifierType="DOI">10.18710/FVHTFM/HX7TBT</relatedIdentifier>
    <relatedIdentifier relationType="HasPart" relatedIdentifierType="DOI">10.18710/FVHTFM/EGURJ1</relatedIdentifier>
    <relatedIdentifier relationType="HasPart" relatedIdentifierType="DOI">10.18710/FVHTFM/NUQXKH</relatedIdentifier>
    <relatedIdentifier relationType="HasPart" relatedIdentifierType="DOI">10.18710/FVHTFM/CJAE1Y</relatedIdentifier>
    <relatedIdentifier relationType="HasPart" relatedIdentifierType="DOI">10.18710/FVHTFM/T6WVMV</relatedIdentifier>
  </relatedIdentifiers>
  <sizes>
    <size>48718</size>
    <size>4972</size>
    <size>50076558</size>
    <size>50076560</size>
    <size>6290</size>
    <size>15220</size>
  </sizes>
  <formats>
    <format>text/tsv</format>
    <format>text/tsv</format>
    <format>text/tsv</format>
    <format>text/tsv</format>
    <format>application/octet-stream</format>
    <format>text/plain</format>
  </formats>
  <version>1.1</version>
  <rightsList>
    <rights rightsURI="info:eu-repo/semantics/openAccess"/>
    <rights rightsURI="https://dataverse.no/api/datasets/:persistentId/versions/1.1/customlicense?persistentId=doi:10.18710/FVHTFM">Custom terms specific to this dataset</rights>
  </rightsList>
  <descriptions>
    <description descriptionType="Abstract">&amp;lt;p&amp;gt;&amp;lt;b&amp;gt;Dataset description&amp;lt;/b&amp;gt;&amp;lt;/p&amp;gt;
&amp;lt;p&amp;gt;This dataset contains background data and supplementary material for Sönning (forthcoming), a study that looks at the behavior of dispersion measures when applied to text-level frequency data. For the literature survey reported in that study, which examines how dispersion measures are used in corpus-based work, it includes tabular files listing the 730 research articles that were examined as well as annotations for those studies that measured dispersion in the corpus-linguistic (and lexicographic) sense. As for the corpus data that were used to train the statistical model parameters underlying the simulation study reported in that paper, the dataset contains a term-document matrix for the 49,604 unique word forms (after conversion to lower-case) that occur in the Brown Corpus. Further, R scripts are included that document in detail how the Brown Corpus XML files, which are available from the Natural Language Toolkit (Bird et al. 2009; https://www.nltk.org/), were processed to produce this data arrangement.&amp;lt;/p&amp;gt;</description>
    <description descriptionType="Abstract">&amp;lt;p&amp;gt;&amp;lt;b&amp;gt;Abstract: Related publication&amp;lt;/b&amp;gt;&amp;lt;/p&amp;gt;
&amp;lt;p&amp;gt;This paper offers a survey of recent corpus-based work, which shows that dispersion is typically measured across the text files in a corpus. Systematic insights into the behavior of measures in such distributional settings are currently lacking, however. After a thorough discussion of six prominent indices, we investigate their behavior on relevant frequency distributions, which are designed to mimic actual corpus data. Our evaluation considers different distributional settings, i.e. various combinations of frequency and dispersion values. The primary focus is on the response of measures to relatively high and low sub-frequencies, i.e. texts in which the item or structure of interest is over- or underrepresented (if not absent). We develop a simple method for constructing sensitivity profiles, which allow us to draw instructive comparisons among measures. We observe that these profiles vary considerably across distributional settings. While D and DP appear to show the most balanced response contours, our findings suggest that much work remains to be done to understand the performance of measures on items with normalized frequencies below 100 per million words.&amp;lt;/p&amp;gt;</description>
    <description descriptionType="TechnicalInfo">MAXQDA Plus, 22.5.0</description>
    <description descriptionType="TechnicalInfo">R, 4.2.1</description>
  </descriptions>
  <geoLocations>
    <geoLocation>
      <geoLocationPlace>United States</geoLocationPlace>
    </geoLocation>
    <geoLocation>
      <geoLocationPlace>Bamberg, Germany</geoLocationPlace>
    </geoLocation>
  </geoLocations>
</resource>
