<?xml version="1.0" encoding="UTF-8"?>
<?xml-model https://raw.githubusercontent.com/i-d-e/ride/master/schema/ride.rng application/xml http://relaxng.org/ns/structure/1.0?>
<?xml-model https://raw.githubusercontent.com/i-d-e/ride/master/schema/ride.rng application/xml http://purl.oclc.org/dsdl/schematron?>
<TEI xmlns="http://www.tei-c.org/ns/1.0" xml:id="ride.6.9">
   <teiHeader>
      <fileDesc>
         <titleStmt>
            <title>«La Repubblica» Corpus</title>
            <author ref="https://orcid.org/0000-0002-5323-4543">
               <name>
                  <forename>Rebecca</forename>
                  <surname>Sierig</surname>
               </name>
               <affiliation>
                  <orgName>University of Leipzig</orgName>
                  <placeName ref="https://www.geonames.org/2879139">Leipzig, Germany</placeName>
               </affiliation>
               <email>rebecca.sierig@uni-leipzig.de</email>
            </author>
         </titleStmt>
         <publicationStmt>
            <publisher>Institut für Dokumentologie und Editorik e.V.</publisher>
            <date when="2017-09">September 2017</date>
            <idno type="URI">https://ride.i-d-e.de/issues/issue-6/la-repubblica-corpus/</idno>
            <idno type="DOI">10.18716/ride.a.6.9</idno>
            <idno type="archive">https://github.com/i-d-e/ride/raw/master/issues/issue06/LaRepubblica/LaRepubblica.pdf</idno>
            <availability>
               <licence target="http://creativecommons.org/licenses/by/4.0/"/>
            </availability>
         </publicationStmt>
         <seriesStmt>
            <title level="j">RIDE - A review journal for digital editions and resources</title>
            <editor ref="https://orcid.org/0000-0003-2852-065X">Ulrike Henny-Krahmer</editor>
            <editor ref="https://orcid.org/0000-0001-8279-9298">Frederike Neuber</editor>
            <editor ref="https://orcid.org/0000-0002-6457-0913" role="managing">Philipp
               Steinkrüger</editor>
            <editor ref="http://viaf.org/viaf/80243768" role="technical">Bernhard Assmann</editor>
            <editor ref="https://orcid.org/0000-0003-2852-065X" role="technical">Ulrike
               Henny-Krahmer</editor>
            <editor ref="https://orcid.org/0000-0001-8279-9298" role="technical">Frederike
               Neuber</editor>
            <biblScope unit="issue" n="6">Digital Text Collections</biblScope>
            <idno type="URI">http://ride.i-d-e.de/issues/issue-6</idno>
            <idno type="DOI">10.18716/ride.a.6</idno>
         </seriesStmt>
         <notesStmt>
            <relatedItem type="reviewed_resource">
               <bibl>
                  <title>«La Repubblica» Corpus</title>
                  <editor>Marco Baroni, Silvia Bernardini, Sara Castagnoli, Federica Comastri,
                     Lorenzo Piccioni, Alessandra Volpi, Guy Aston, Marco Mazzoleni, Eros
                     Zanchetta</editor>
                  <respStmt>
                     <resp>Contributor</resp>
                     <persName>Marco Baroni</persName>
                  </respStmt>
                  <respStmt>
                     <resp>Contributor</resp>
                     <persName>Silvia Bernardini</persName>
                  </respStmt>
                  <respStmt>
                     <resp>Contributor</resp>
                     <persName>Sara Castagnoli</persName>
                  </respStmt>
                  <respStmt>
                     <resp>Contributor</resp>
                     <persName>Federica Comastri</persName>
                  </respStmt>
                  <respStmt>
                     <resp>Contributor</resp>
                     <persName>Lorenzo Piccioni</persName>
                  </respStmt>
                  <respStmt>
                     <resp>Contributor</resp>
                     <persName>Alessandra Volpi</persName>
                  </respStmt>
                  <respStmt>
                     <resp>Contributor</resp>
                     <persName>Guy Aston</persName>
                  </respStmt>
                  <respStmt>
                     <resp>Contributor</resp>
                     <persName>Marco Mazzoleni</persName>
                  </respStmt>
                  <respStmt>
                     <resp>Contributor</resp>
                     <persName>Eros Zanchetta</persName>
                  </respStmt>
                  <date type="publication">2004ff.</date>
                  <idno type="URI">https://corpora.dipintra.it/public/run.cgi/first?corpname=repubblica</idno>
                  <date type="accessed">2017-08-01</date>
               </bibl>
            </relatedItem>
            <relatedItem type="reviewing_criteria">
               <bibl>
                  <ref target="http://www.i-d-e.de/criteria-text-collections-version-1-0">Criteria
                     for Reviewing Digital Text Collections, version 1.0</ref>
               </bibl>
            </relatedItem>
         </notesStmt>
         <sourceDesc>
            <p>born digital</p>
         </sourceDesc>
      </fileDesc>
      <encodingDesc>
         <classDecl>
            <taxonomy xml:base="http://www.i-d-e.de/criteria-text-collections-version-1-0">
               <category xml:id="general_information">
                  <category xml:id="bibl_desc">
                     <catDesc>
                        <ref target="#K1.1">cf. Catalogue 1.1</ref>
                     </catDesc>
                     <catDesc>Can the text collection be identified in terms similar to traditional
                        bibliographic descriptions (title, responsible editors, institution, date(s)
                        of publication, identifier/address)?</catDesc>
                     <catDesc>
                        <num type="boolean" value="1"/>
                     </catDesc>
                  </category>
                  <category xml:id="contributors">
                     <catDesc>
                        <ref target="#K1.3">cf. Catalogue 1.3</ref>
                     </catDesc>
                     <catDesc>Are the contributors (editors, institutions, associates) of the
                        project documented?</catDesc>
                     <catDesc>
                        <num type="boolean" value="1"/>
                     </catDesc>
                  </category>
                  <category xml:id="contacts">
                     <catDesc>
                        <ref target="#K1.4">cf. Catalogue 1.4</ref>
                     </catDesc>
                     <catDesc>Is contact information given?</catDesc>
                     <catDesc>
                        <num type="boolean" value="1"/>
                     </catDesc>
                  </category>
               </category>
               <category xml:id="aims">
                  <category xml:id="doc_contents">
                     <catDesc>
                        <ref target="#K2.1">cf. Catalogue 2.1</ref>
                     </catDesc>
                     <catDesc>Is there a description of the aims and contents of the text
                        collection?</catDesc>
                     <catDesc>
                        <num type="boolean" value="1"/>
                     </catDesc>
                  </category>
                  <category xml:id="purpose">
                     <catDesc>
                        <ref target="#K2.2">cf. Catalogue 2.2</ref>
                     </catDesc>
                     <catDesc>What is the purpose of the text collection?</catDesc>
                     <category xml:id="research">
                        <catDesc>
                           <num type="boolean" value="1"/>
                        </catDesc>
                     </category>
                     <category xml:id="teaching">
                        <catDesc>
                           <num type="boolean" value="1"/>
                        </catDesc>
                     </category>
                     <category xml:id="general_purpose">
                        <catDesc>
                           <num type="boolean" value="0"/>
                        </catDesc>
                     </category>
                     <category corresp="#other">
                        <catDesc>
                           <num type="boolean" value="0"/>
                        </catDesc>
                        <category corresp="#free">
                           <desc/>
                        </category>
                     </category>
                  </category>
                  <category xml:id="research_type">
                     <catDesc>
                        <ref target="#K3.1.8">cf. Catalogue 3.1.8</ref>
                     </catDesc>
                     <catDesc>What kind of research does the collection allow to conduct
                        primarily?</catDesc>
                     <category xml:id="qualitative">
                        <catDesc>
                           <num type="boolean" value="1"/>
                        </catDesc>
                     </category>
                     <category xml:id="quantitative">
                        <catDesc>
                           <num type="boolean" value="0"/>
                        </catDesc>
                     </category>
                     <category corresp="#none">
                        <catDesc>
                           <num type="boolean" value="0"/>
                        </catDesc>
                     </category>
                  </category>
                  <category xml:id="classification">
                     <catDesc>
                        <ref target="#K2.3">cf. Catalogue 2.3</ref>
                     </catDesc>
                     <catDesc>How does the text collection classify itself (e.g. in its title or
                        documentation)?</catDesc>
                     <category xml:id="collection">
                        <catDesc>
                           <num type="boolean" value="0"/>
                        </catDesc>
                     </category>
                     <category xml:id="corpus">
                        <catDesc>
                           <num type="boolean" value="1"/>
                        </catDesc>
                     </category>
                     <category xml:id="digital_archive">
                        <catDesc>
                           <num type="boolean" value="0"/>
                        </catDesc>
                     </category>
                     <category xml:id="digital_library">
                        <catDesc>
                           <num type="boolean" value="0"/>
                        </catDesc>
                     </category>
                     <category xml:id="digital_edition">
                        <catDesc>
                           <num type="boolean" value="0"/>
                        </catDesc>
                     </category>
                     <category xml:id="portal">
                        <catDesc>
                           <num type="boolean" value="0"/>
                        </catDesc>
                     </category>
                     <category xml:id="database">
                        <catDesc>
                           <num type="boolean" value="0"/>
                        </catDesc>
                     </category>
                     <category xml:id="no_classification">
                        <catDesc>
                           <num type="boolean" value="0"/>
                        </catDesc>
                     </category>
                     <category corresp="#other">
                        <catDesc>
                           <num type="boolean" value="0"/>
                        </catDesc>
                        <category corresp="#free">
                           <desc/>
                        </category>
                     </category>
                  </category>
                  <category xml:id="research_fields">
                     <catDesc>
                        <ref target="#K2.2">cf. Catalogue 2.2</ref>
                     </catDesc>
                     <catDesc>To which field(s) of research does the text collection
                        contribute?</catDesc>
                     <category xml:id="field_history">
                        <catDesc>
                           <num type="boolean" value="0"/>
                        </catDesc>
                     </category>
                     <category xml:id="field_literary_studies">
                        <catDesc>
                           <num type="boolean" value="0"/>
                        </catDesc>
                     </category>
                     <category xml:id="field_linguistics">
                        <catDesc>
                           <num type="boolean" value="1"/>
                        </catDesc>
                     </category>
                     <category xml:id="field_musicology">
                        <catDesc>
                           <num type="boolean" value="0"/>
                        </catDesc>
                     </category>
                     <category xml:id="field_art_history">
                        <catDesc>
                           <num type="boolean" value="0"/>
                        </catDesc>
                     </category>
                     <category xml:id="field_archaeology">
                        <catDesc>
                           <num type="boolean" value="0"/>
                        </catDesc>
                     </category>
                     <category xml:id="field_philosophy">
                        <catDesc>
                           <num type="boolean" value="0"/>
                        </catDesc>
                     </category>
                     <category xml:id="field_religious_studies">
                        <catDesc>
                           <num type="boolean" value="0"/>
                        </catDesc>
                     </category>
                     <category xml:id="field_sociology">
                        <catDesc>
                           <num type="boolean" value="0"/>
                        </catDesc>
                     </category>
                     <category corresp="#other">
                        <catDesc>
                           <num type="boolean" value="1"/>
                        </catDesc>
                        <category corresp="#free">
                           <desc>Translational Studies</desc>
                        </category>
                     </category>
                     <category corresp="#none">
                        <catDesc>
                           <num type="boolean" value="0"/>
                        </catDesc>
                     </category>
                  </category>
               </category>
               <category xml:id="content">
                  <category xml:id="era">
                     <catDesc>
                        <ref target="#K2.5">cf. Catalogue 2.5</ref>
                     </catDesc>
                     <catDesc>What era(s) do the texts belong to?</catDesc>
                     <category xml:id="era_classics">
                        <catDesc>
                           <num type="boolean" value="0"/>
                        </catDesc>
                     </category>
                     <category xml:id="era_medieval">
                        <catDesc>
                           <num type="boolean" value="0"/>
                        </catDesc>
                     </category>
                     <category xml:id="era_early_modern">
                        <catDesc>
                           <num type="boolean" value="0"/>
                        </catDesc>
                     </category>
                     <category xml:id="era_modern">
                        <catDesc>
                           <num type="boolean" value="0"/>
                        </catDesc>
                     </category>
                     <category xml:id="era_contemporary">
                        <catDesc>
                           <num type="boolean" value="1"/>
                        </catDesc>
                     </category>
                  </category>
                  <category xml:id="language">
                     <catDesc>
                        <ref target="#K2.5">cf. Catalogue 2.5</ref>
                     </catDesc>
                     <catDesc>What languages are the texts in?</catDesc>
                     <category xml:id="arabic">
                        <catDesc>
                           <num type="boolean" value="0"/>
                        </catDesc>
                     </category>
                     <category xml:id="chinese">
                        <catDesc>
                           <num type="boolean" value="0"/>
                        </catDesc>
                     </category>
                     <category xml:id="danish">
                        <catDesc>
                           <num type="boolean" value="0"/>
                        </catDesc>
                     </category>
                     <category xml:id="english">
                        <catDesc>
                           <num type="boolean" value="0"/>
                        </catDesc>
                     </category>
                     <category xml:id="finnish">
                        <catDesc>
                           <num type="boolean" value="0"/>
                        </catDesc>
                     </category>
                     <category xml:id="french">
                        <catDesc>
                           <num type="boolean" value="0"/>
                        </catDesc>
                     </category>
                     <category xml:id="german">
                        <catDesc>
                           <num type="boolean" value="0"/>
                        </catDesc>
                     </category>
                     <category xml:id="greek">
                        <catDesc>
                           <num type="boolean" value="0"/>
                        </catDesc>
                     </category>
                     <category xml:id="hebrew">
                        <catDesc>
                           <num type="boolean" value="0"/>
                        </catDesc>
                     </category>
                     <category xml:id="hindi">
                        <catDesc>
                           <num type="boolean" value="0"/>
                        </catDesc>
                     </category>
                     <category xml:id="italian">
                        <catDesc>
                           <num type="boolean" value="1"/>
                        </catDesc>
                     </category>
                     <category xml:id="japanese">
                        <catDesc>
                           <num type="boolean" value="0"/>
                        </catDesc>
                     </category>
                     <category xml:id="latin">
                        <catDesc>
                           <num type="boolean" value="0"/>
                        </catDesc>
                     </category>
                     <category xml:id="norwegian">
                        <catDesc>
                           <num type="boolean" value="0"/>
                        </catDesc>
                     </category>
                     <category xml:id="polish">
                        <catDesc>
                           <num type="boolean" value="0"/>
                        </catDesc>
                     </category>
                     <category xml:id="portuguese">
                        <catDesc>
                           <num type="boolean" value="0"/>
                        </catDesc>
                     </category>
                     <category xml:id="russian">
                        <catDesc>
                           <num type="boolean" value="0"/>
                        </catDesc>
                     </category>
                     <category xml:id="spanish">
                        <catDesc>
                           <num type="boolean" value="0"/>
                        </catDesc>
                     </category>
                     <category xml:id="swedish">
                        <catDesc>
                           <num type="boolean" value="0"/>
                        </catDesc>
                     </category>
                     <category xml:id="turkish">
                        <catDesc>
                           <num type="boolean" value="0"/>
                        </catDesc>
                     </category>
                     <category corresp="#other">
                        <catDesc>
                           <num type="boolean" value="0"/>
                        </catDesc>
                        <category corresp="#free">
                           <desc/>
                        </category>
                     </category>
                     <category corresp="#free" xml:id="language_note">
                        <desc/>
                     </category>
                  </category>
                  <category xml:id="text_type">
                     <catDesc>
                        <ref target="#K2.5">cf. Catalogue 2.5</ref>
                     </catDesc>
                     <catDesc>What kind of texts are in the collection?</catDesc>
                     <category xml:id="literary_works">
                        <catDesc>
                           <num type="boolean" value="0"/>
                        </catDesc>
                     </category>
                     <category xml:id="private_documents">
                        <catDesc>
                           <num type="boolean" value="0"/>
                        </catDesc>
                     </category>
                     <category xml:id="essays">
                        <catDesc>
                           <num type="boolean" value="0"/>
                        </catDesc>
                     </category>
                     <category xml:id="newspaper_articles">
                        <catDesc>
                           <num type="boolean" value="1"/>
                        </catDesc>
                     </category>
                     <category xml:id="charters">
                        <catDesc>
                           <num type="boolean" value="0"/>
                        </catDesc>
                     </category>
                     <category xml:id="inscriptions">
                        <catDesc>
                           <num type="boolean" value="0"/>
                        </catDesc>
                     </category>
                     <category xml:id="files_records">
                        <catDesc>
                           <num type="boolean" value="0"/>
                        </catDesc>
                     </category>
                     <category xml:id="protocols">
                        <catDesc>
                           <num type="boolean" value="0"/>
                        </catDesc>
                     </category>
                     <category xml:id="scientific_papers">
                        <catDesc>
                           <num type="boolean" value="0"/>
                        </catDesc>
                     </category>
                     <category xml:id="speech_transcripts">
                        <catDesc>
                           <num type="boolean" value="0"/>
                        </catDesc>
                     </category>
                     <category corresp="#other">
                        <catDesc>
                           <num type="boolean" value="0"/>
                        </catDesc>
                        <category corresp="#free">
                           <desc/>
                        </category>
                     </category>
                  </category>
                  <category xml:id="add_information">
                     <catDesc>
                        <ref target="#K2.5">cf. Catalogue 2.5</ref>
                     </catDesc>
                     <catDesc>What kind of information is published in addition to the
                        texts?</catDesc>
                     <category xml:id="introduction">
                        <catDesc>
                           <num type="boolean" value="0"/>
                        </catDesc>
                     </category>
                     <category xml:id="commentary">
                        <catDesc>
                           <num type="boolean" value="0"/>
                        </catDesc>
                     </category>
                     <category xml:id="context_material">
                        <catDesc>
                           <num type="boolean" value="1"/>
                        </catDesc>
                     </category>
                     <category xml:id="bibliography">
                        <catDesc>
                           <num type="boolean" value="0"/>
                        </catDesc>
                     </category>
                     <category xml:id="facsimile">
                        <catDesc>
                           <num type="boolean" value="0"/>
                        </catDesc>
                     </category>
                     <category corresp="#other">
                        <catDesc>
                           <num type="boolean" value="1"/>
                        </catDesc>
                        <category corresp="#free">
                           <desc/>
                        </category>
                     </category>
                     <category corresp="#none">
                        <catDesc>
                           <num type="boolean" value="0"/>
                        </catDesc>
                     </category>
                  </category>
               </category>
               <category xml:id="composition">
                  <category xml:id="documentation_methods">
                     <catDesc>
                        <ref target="#K3.1.1">cf. Catalogue 3.1.1-3.1.3</ref>
                     </catDesc>
                     <catDesc>Are the principles and decisions regarding the design of the text
                        collection, its composition and the selection of texts documented?</catDesc>
                     <catDesc>
                        <num type="boolean" value="1"/>
                     </catDesc>
                  </category>
                  <category xml:id="selection">
                     <catDesc>
                        <ref target="#K3.1">cf. Catalogue 3.1</ref>
                     </catDesc>
                     <catDesc>What selection criteria have been chosen for the text
                        collection?</catDesc>
                     <category xml:id="selection_language">
                        <catDesc>
                           <num type="boolean" value="1"/>
                        </catDesc>
                     </category>
                     <category xml:id="selection_author">
                        <catDesc>
                           <num type="boolean" value="0"/>
                        </catDesc>
                     </category>
                     <category xml:id="selection_country">
                        <catDesc>
                           <num type="boolean" value="0"/>
                        </catDesc>
                     </category>
                     <category xml:id="selection_epoch">
                        <catDesc>
                           <num type="boolean" value="1"/>
                        </catDesc>
                     </category>
                     <category xml:id="selection_genre">
                        <catDesc>
                           <num type="boolean" value="1"/>
                        </catDesc>
                     </category>
                     <category xml:id="selection_topic">
                        <catDesc>
                           <num type="boolean" value="0"/>
                        </catDesc>
                     </category>
                     <category xml:id="selection_style">
                        <catDesc>
                           <num type="boolean" value="0"/>
                        </catDesc>
                     </category>
                     <category xml:id="selection_linguistic_characteristics">
                        <catDesc>
                           <num type="boolean" value="0"/>
                        </catDesc>
                     </category>
                     <category corresp="#other">
                        <catDesc>
                           <num type="boolean" value="0"/>
                        </catDesc>
                        <category corresp="#free">
                           <desc/>
                        </category>
                     </category>
                  </category>
                  <category xml:id="size">
                     <category xml:id="size_texts">
                        <catDesc>
                           <ref target="#K3.1.4">cf. Catalogue 3.1.4</ref>
                        </catDesc>
                        <catDesc>How large is the text collection in number of
                           texts/records?</catDesc>
                        <category xml:id="texts_le10">
                           <catDesc>
                              <num type="boolean" value="0"/>
                           </catDesc>
                        </category>
                        <category xml:id="texts_11-50">
                           <catDesc>
                              <num type="boolean" value="0"/>
                           </catDesc>
                        </category>
                        <category xml:id="texts_51-100">
                           <catDesc>
                              <num type="boolean" value="0"/>
                           </catDesc>
                        </category>
                        <category xml:id="texts_gt100_">
                           <catDesc>
                              <num type="boolean" value="0"/>
                           </catDesc>
                        </category>
                        <category xml:id="texts_gt1000">
                           <catDesc>
                              <num type="boolean" value="1"/>
                           </catDesc>
                        </category>
                        <category corresp="#unknown">
                           <catDesc>
                              <num type="boolean" value="0"/>
                           </catDesc>
                        </category>
                     </category>
                     <category xml:id="size_tokens">
                        <catDesc>
                           <ref target="#K3.1.4">cf. Catalogue 3.1.4</ref>
                        </catDesc>
                        <catDesc>How large is the text collection in number of tokens?</catDesc>
                        <category xml:id="tokens_lt100.000">
                           <catDesc>
                              <num type="boolean" value="0"/>
                           </catDesc>
                        </category>
                        <category xml:id="tokens_100.000-1mio">
                           <catDesc>
                              <num type="boolean" value="0"/>
                           </catDesc>
                        </category>
                        <category xml:id="tokens_gt1mio">
                           <catDesc>
                              <num type="boolean" value="0"/>
                           </catDesc>
                        </category>
                        <category xml:id="tokens_gt10mio">
                           <catDesc>
                              <num type="boolean" value="1"/>
                           </catDesc>
                        </category>
                        <category corresp="#unknown">
                           <catDesc>
                              <num type="boolean" value="0"/>
                           </catDesc>
                        </category>
                     </category>
                  </category>
                  <category xml:id="structure">
                     <catDesc>
                        <ref target="#K3.1.5">cf. Catalogue 3.1.5</ref>
                     </catDesc>
                     <catDesc>Does the text collection have identifiable sub-collections or
                        components?</catDesc>
                     <catDesc>
                        <num type="boolean" value="1"/>
                     </catDesc>
                  </category>
                  <category xml:id="acquisition">
                     <category xml:id="text_recording">
                        <catDesc>
                           <ref target="#K3.1.6">cf. Catalogue 3.1.6</ref>
                        </catDesc>
                        <catDesc>Does the text collection record or transcribe the textual data for
                           the first time?</catDesc>
                        <catDesc>
                           <num type="boolean" value="1"/>
                        </catDesc>
                     </category>
                     <category xml:id="text_integration">
                        <catDesc>
                           <ref target="#K3.1.6">cf. Catalogue 3.1.6</ref>
                        </catDesc>
                        <catDesc>What kind of material has been taken over from other
                           sources?</catDesc>
                        <category xml:id="full_text">
                           <catDesc>
                              <num type="boolean" value="1"/>
                           </catDesc>
                        </category>
                        <category xml:id="reuse_metadata">
                           <catDesc>
                              <num type="boolean" value="1"/>
                           </catDesc>
                        </category>
                        <category xml:id="reuse_annotation">
                           <catDesc>
                              <num type="boolean" value="0"/>
                           </catDesc>
                        </category>
                        <category corresp="#other">
                           <catDesc>
                              <num type="boolean" value="0"/>
                           </catDesc>
                           <category corresp="#free">
                              <desc/>
                           </category>
                        </category>
                        <category corresp="#none">
                           <catDesc>
                              <num type="boolean" value="0"/>
                           </catDesc>
                        </category>
                     </category>
                  </category>
                  <category xml:id="quality">
                     <catDesc>
                        <ref target="#K3.1.7">cf. Catalogue 3.1.7</ref>
                     </catDesc>
                     <catDesc>Has the quality of the data (transcriptions, metadata, annotations,
                        etc.) been checked?</catDesc>
                     <catDesc>
                        <num type="boolean" value="1"/>
                     </catDesc>
                  </category>
                  <category xml:id="typology">
                     <catDesc>
                        <ref target="#K3.1.8">cf. Catalogue 3.1.8</ref>
                     </catDesc>
                     <catDesc>Considering aims and methods of the text collection, how would you
                        classify it further? For definitions please consider the
                        help-texts.</catDesc>
                     <category xml:id="typology_general_purpose">
                        <catDesc>
                           <num type="boolean" value="0"/>
                        </catDesc>
                     </category>
                     <category xml:id="typology_corpus">
                        <catDesc>
                           <num type="boolean" value="1"/>
                        </catDesc>
                     </category>
                     <category xml:id="typology_collection_records">
                        <catDesc>
                           <num type="boolean" value="0"/>
                        </catDesc>
                     </category>
                     <category xml:id="typology_canon">
                        <catDesc>
                           <num type="boolean" value="0"/>
                        </catDesc>
                     </category>
                     <category xml:id="typology_oeuvre">
                        <catDesc>
                           <num type="boolean" value="0"/>
                        </catDesc>
                     </category>
                     <category xml:id="typology_reference_corpus">
                        <catDesc>
                           <num type="boolean" value="0"/>
                        </catDesc>
                     </category>
                     <category xml:id="typology_contrastive_corpus">
                        <catDesc>
                           <num type="boolean" value="0"/>
                        </catDesc>
                     </category>
                     <category xml:id="typology_parallel_corpus">
                        <catDesc>
                           <num type="boolean" value="0"/>
                        </catDesc>
                     </category>
                     <category xml:id="typology_diachronic_corpus">
                        <catDesc>
                           <num type="boolean" value="0"/>
                        </catDesc>
                     </category>
                     <category corresp="#other">
                        <catDesc>
                           <num type="boolean" value="0"/>
                        </catDesc>
                        <category corresp="#free">
                           <desc/>
                        </category>
                     </category>
                  </category>
               </category>
               <category xml:id="data_modelling">
                  <category xml:id="text_treatment">
                     <catDesc>
                        <ref target="#K3.2.1">cf. Catalogue 3.2.1</ref>
                     </catDesc>
                     <catDesc>How are the textual sources represented in the digital
                        collection?</catDesc>
                     <category xml:id="normalized_transcription">
                        <catDesc>
                           <num type="boolean" value="1"/>
                        </catDesc>
                     </category>
                     <category xml:id="orthographic_transcription">
                        <catDesc>
                           <num type="boolean" value="0"/>
                        </catDesc>
                     </category>
                     <category xml:id="phonetic_transcription">
                        <catDesc>
                           <num type="boolean" value="0"/>
                        </catDesc>
                     </category>
                     <category xml:id="diplomatic_transcription">
                        <catDesc>
                           <num type="boolean" value="0"/>
                        </catDesc>
                     </category>
                     <category xml:id="transliteration">
                        <catDesc>
                           <num type="boolean" value="0"/>
                        </catDesc>
                     </category>
                     <category xml:id="edited_text">
                        <catDesc>
                           <num type="boolean" value="0"/>
                        </catDesc>
                     </category>
                     <category xml:id="translated_text">
                        <catDesc>
                           <num type="boolean" value="0"/>
                        </catDesc>
                     </category>
                     <category xml:id="summarized_text">
                        <catDesc>
                           <num type="boolean" value="0"/>
                        </catDesc>
                     </category>
                     <category xml:id="sampled_text">
                        <catDesc>
                           <num type="boolean" value="0"/>
                        </catDesc>
                     </category>
                     <category corresp="#other">
                        <catDesc>
                           <num type="boolean" value="0"/>
                        </catDesc>
                        <category corresp="#free">
                           <desc/>
                        </category>
                     </category>
                  </category>
                  <category xml:id="basic_format">
                     <catDesc>
                        <ref target="#K3.2.4">cf. Catalogue 3.2.4</ref>
                     </catDesc>
                     <catDesc>In which basic format are the texts encoded?</catDesc>
                     <category xml:id="plain_text">
                        <catDesc>
                           <num type="boolean" value="0"/>
                        </catDesc>
                     </category>
                     <category xml:id="xml">
                        <catDesc>
                           <num type="boolean" value="1"/>
                        </catDesc>
                     </category>
                     <category xml:id="html">
                        <catDesc>
                           <num type="boolean" value="0"/>
                        </catDesc>
                     </category>
                     <category corresp="#other">
                        <catDesc>
                           <num type="boolean" value="0"/>
                        </catDesc>
                        <category corresp="#free">
                           <desc/>
                        </category>
                     </category>
                  </category>
                  <category xml:id="annotation">
                     <category xml:id="annotation_type">
                        <catDesc>
                           <ref target="#K3.2.2">cf. Catalogue 3.2.2</ref>
                        </catDesc>
                        <catDesc>With what information are the texts further enriched?</catDesc>
                        <category xml:id="semantic_annotations">
                           <catDesc>
                              <num type="boolean" value="1"/>
                           </catDesc>
                        </category>
                        <category xml:id="linguistic_annotations">
                           <catDesc>
                              <num type="boolean" value="1"/>
                           </catDesc>
                        </category>
                        <category xml:id="editorial_annotations">
                           <catDesc>
                              <num type="boolean" value="0"/>
                           </catDesc>
                        </category>
                        <category xml:id="structural_information">
                           <catDesc>
                              <num type="boolean" value="1"/>
                           </catDesc>
                        </category>
                        <category corresp="#other">
                           <catDesc>
                              <num type="boolean" value="0"/>
                           </catDesc>
                           <category corresp="#free">
                              <desc/>
                           </category>
                        </category>
                        <category corresp="#none">
                           <catDesc>
                              <num type="boolean" value="0"/>
                           </catDesc>
                        </category>
                     </category>
                     <category xml:id="annotation_integration">
                        <catDesc>
                           <ref target="#K3.2.2">cf. Catalogue 3.2.2</ref>
                        </catDesc>
                        <catDesc>How are the annotations linked to the texts themselves?</catDesc>
                        <category xml:id="embedded">
                           <catDesc>
                              <num type="boolean" value="0"/>
                           </catDesc>
                        </category>
                        <category xml:id="stand-off">
                           <catDesc>
                              <num type="boolean" value="0"/>
                           </catDesc>
                        </category>
                        <category corresp="#other">
                           <catDesc>
                              <num type="boolean" value="0"/>
                           </catDesc>
                           <category corresp="#free">
                              <desc/>
                           </category>
                        </category>
                        <category corresp="#not_applicable">
                           <catDesc>
                              <num type="boolean" value="1"/>
                           </catDesc>
                        </category>
                     </category>
                  </category>
                  <category xml:id="metadata">
                     <category xml:id="metadata_type">
                        <catDesc>
                           <ref target="#K3.2.3">cf. Catalogue 3.2.3</ref>
                        </catDesc>
                        <catDesc>What kind of metadata are included in the text
                           collection?</catDesc>
                        <category xml:id="descriptive">
                           <catDesc>
                              <num type="boolean" value="1"/>
                           </catDesc>
                        </category>
                        <category xml:id="structural">
                           <catDesc>
                              <num type="boolean" value="1"/>
                           </catDesc>
                        </category>
                        <category xml:id="administrative">
                           <catDesc>
                              <num type="boolean" value="0"/>
                           </catDesc>
                        </category>
                        <category corresp="#other">
                           <catDesc>
                              <num type="boolean" value="0"/>
                           </catDesc>
                           <category corresp="#free">
                              <desc/>
                           </category>
                        </category>
                        <category corresp="#none">
                           <catDesc>
                              <num type="boolean" value="0"/>
                           </catDesc>
                        </category>
                     </category>
                     <category xml:id="metadata_level">
                        <catDesc>
                           <ref target="#K3.2.2">cf. Catalogue 3.2.2</ref>
                        </catDesc>
                        <catDesc>On which level are the metadata included?</catDesc>
                        <category xml:id="whole_collection">
                           <catDesc>
                              <num type="boolean" value="1"/>
                           </catDesc>
                        </category>
                        <category xml:id="collection_parts">
                           <catDesc>
                              <num type="boolean" value="0"/>
                           </catDesc>
                        </category>
                        <category xml:id="individual_texts">
                           <catDesc>
                              <num type="boolean" value="1"/>
                           </catDesc>
                        </category>
                        <category corresp="#other">
                           <catDesc>
                              <num type="boolean" value="0"/>
                           </catDesc>
                           <category corresp="#free">
                              <desc/>
                           </category>
                        </category>
                        <category corresp="#not_applicable">
                           <catDesc>
                              <num type="boolean" value="0"/>
                           </catDesc>
                        </category>
                     </category>
                  </category>
                  <category xml:id="data_standards">
                     <category xml:id="data_schema">
                        <catDesc>
                           <ref target="#K3.2.4">cf. Catalogue 3.2.4</ref>
                        </catDesc>
                        <catDesc>What kind of data/metadata/annotation schemas are used for the text
                           collection?</catDesc>
                        <category xml:id="standardized_schema">
                           <catDesc>
                              <num type="boolean" value="0"/>
                           </catDesc>
                        </category>
                        <category xml:id="customized_standard_schema">
                           <catDesc>
                              <num type="boolean" value="0"/>
                           </catDesc>
                        </category>
                        <category xml:id="project_specific">
                           <catDesc>
                              <num type="boolean" value="0"/>
                           </catDesc>
                        </category>
                        <category corresp="#none">
                           <catDesc>
                              <num type="boolean" value="0"/>
                           </catDesc>
                        </category>
                        <category corresp="#unknown">
                           <catDesc>
                              <num type="boolean" value="1"/>
                           </catDesc>
                        </category>
                     </category>
                     <category xml:id="standard_format">
                        <catDesc>
                           <ref target="#K3.2.4">cf. Catalogue 3.2.4</ref>
                        </catDesc>
                        <catDesc>Which standards for text encoding, metadata and annotation are used
                           in the text collection?</catDesc>
                        <category xml:id="tei">
                           <catDesc>
                              <num type="boolean" value="1"/>
                           </catDesc>
                        </category>
                        <category xml:id="cei">
                           <catDesc>
                              <num type="boolean" value="0"/>
                           </catDesc>
                        </category>
                        <category xml:id="ead">
                           <catDesc>
                              <num type="boolean" value="0"/>
                           </catDesc>
                        </category>
                        <category xml:id="xces">
                           <catDesc>
                              <num type="boolean" value="0"/>
                           </catDesc>
                        </category>
                        <category xml:id="dublin_core">
                           <catDesc>
                              <num type="boolean" value="0"/>
                           </catDesc>
                        </category>
                        <category xml:id="edm">
                           <catDesc>
                              <num type="boolean" value="0"/>
                           </catDesc>
                        </category>
                        <category xml:id="mets">
                           <catDesc>
                              <num type="boolean" value="0"/>
                           </catDesc>
                        </category>
                        <category xml:id="mods">
                           <catDesc>
                              <num type="boolean" value="0"/>
                           </catDesc>
                        </category>
                        <category xml:id="skos">
                           <catDesc>
                              <num type="boolean" value="0"/>
                           </catDesc>
                        </category>
                        <category xml:id="owl">
                           <catDesc>
                              <num type="boolean" value="0"/>
                           </catDesc>
                        </category>
                        <category xml:id="imdi">
                           <catDesc>
                              <num type="boolean" value="0"/>
                           </catDesc>
                        </category>
                        <category xml:id="cmdi">
                           <catDesc>
                              <num type="boolean" value="0"/>
                           </catDesc>
                        </category>
                        <category xml:id="tcf">
                           <catDesc>
                              <num type="boolean" value="0"/>
                           </catDesc>
                        </category>
                        <category xml:id="olac">
                           <catDesc>
                              <num type="boolean" value="0"/>
                           </catDesc>
                        </category>
                        <category xml:id="eagles">
                           <catDesc>
                              <num type="boolean" value="1"/>
                           </catDesc>
                        </category>
                        <category xml:id="pos_tagsets">
                           <catDesc>
                              <num type="boolean" value="0"/>
                           </catDesc>
                        </category>
                        <category corresp="#other">
                           <catDesc>
                              <num type="boolean" value="0"/>
                           </catDesc>
                           <category corresp="#free">
                              <desc/>
                           </category>
                        </category>
                        <category corresp="#none">
                           <catDesc>
                              <num type="boolean" value="0"/>
                           </catDesc>
                        </category>
                     </category>
                  </category>
               </category>
               <category xml:id="provision">
                  <category xml:id="basic_data_accessible">
                     <catDesc>
                        <ref target="#K4.1">cf. Catalogue 4.1</ref>
                     </catDesc>
                     <catDesc>Is the textual data accessible in a source format (e.g. XML,
                        TXT)?</catDesc>
                     <catDesc>
                        <num type="boolean" value="0"/>
                     </catDesc>
                  </category>
                  <category xml:id="download">
                     <catDesc>
                        <ref target="#K4.2">cf. Catalogue 4.2</ref>
                     </catDesc>
                     <catDesc>Can the entire raw data of the project be downloaded (as a
                        whole)?</catDesc>
                     <catDesc>
                        <num type="boolean" value="0"/>
                     </catDesc>
                  </category>
                  <category xml:id="technical_interfaces">
                     <catDesc>
                        <ref target="#K4.2">cf. Catalogue 4.2</ref>
                     </catDesc>
                     <catDesc>Are there technical interfaces which allow the reuse of the data of
                        the text collection in other contexts?</catDesc>
                     <category xml:id="OAI-PMH">
                        <catDesc>
                           <num type="boolean" value="0"/>
                        </catDesc>
                     </category>
                     <category xml:id="REST">
                        <catDesc>
                           <num type="boolean" value="0"/>
                        </catDesc>
                     </category>
                     <category xml:id="SPARQL_endpoint">
                        <catDesc>
                           <num type="boolean" value="0"/>
                        </catDesc>
                     </category>
                     <category xml:id="general_API">
                        <catDesc>
                           <num type="boolean" value="0"/>
                        </catDesc>
                     </category>
                     <category corresp="#other">
                        <catDesc>
                           <num type="boolean" value="0"/>
                        </catDesc>
                        <category corresp="#free">
                           <desc/>
                        </category>
                     </category>
                     <category corresp="#none">
                        <catDesc>
                           <num type="boolean" value="1"/>
                        </catDesc>
                     </category>
                  </category>
                  <category xml:id="analytical_data">
                     <catDesc>
                        <ref target="#K4.3">cf. Catalogue 4.3</ref>
                     </catDesc>
                     <catDesc>Besides the textual data, does the project provide analytical data
                        (e.g. statistics) to download or harvest?</catDesc>
                     <catDesc>
                        <num type="boolean" value="1"/>
                     </catDesc>
                  </category>
                  <category xml:id="reuse">
                     <catDesc>
                        <ref target="#K4.4">cf. Catalogue 4.4</ref>
                     </catDesc>
                     <catDesc>Can you use the data with other tools useful for this kind of
                        content?</catDesc>
                     <catDesc>
                        <num type="boolean" value="1"/>
                     </catDesc>
                  </category>
               </category>
               <category xml:id="user_interface">
                  <category xml:id="interface_provision">
                     <catDesc>
                        <ref target="#K5.1">cf. Catalogue 5.1</ref>
                     </catDesc>
                     <catDesc>Does the text collection have a dedicated user interface designed for
                        the collection at hand in which the texts of the collection are represented
                        and/or in which the data is analyzable?</catDesc>
                     <catDesc>
                        <num type="boolean" value="1"/>
                     </catDesc>
                  </category>
                  <category xml:id="user_interface_sub">
                     <category xml:id="usability">
                        <catDesc>
                           <ref target="#K5.3">cf. Catalogue 5.3</ref>
                        </catDesc>
                        <catDesc>From your point of view, is the interface of the text collection
                           clearly arranged and easy to navigate so that the user can quickly
                           identify the purpose, the content and the main access methods of the
                           resource?</catDesc>
                        <catDesc>
                           <num type="boolean" value="0"/>
                        </catDesc>
                     </category>
                     <category xml:id="access_modes">
                        <category xml:id="browsing">
                           <catDesc>
                              <ref target="#K5.4">cf. Catalogue 5.4</ref>
                           </catDesc>
                           <catDesc>Does the project offer the possibility to browse the contents by
                              simple browsing options or advanced structured access via indices
                              (e.g. by author, year, genre)?</catDesc>
                           <catDesc>
                              <num type="boolean" value="1"/>
                           </catDesc>
                        </category>
                        <category xml:id="full_text_search">
                           <catDesc>
                              <ref target="#K5.4">cf. Catalogue 5.4</ref>
                           </catDesc>
                           <catDesc>Does the project offer a fulltext search?</catDesc>
                           <catDesc>
                              <num type="boolean" value="0"/>
                           </catDesc>
                        </category>
                        <category xml:id="advanced_search">
                           <catDesc>
                              <ref target="#K5.4">cf. Catalogue 5.4</ref>
                           </catDesc>
                           <catDesc>Does the project offer an advanced search?</catDesc>
                           <catDesc>
                              <num type="boolean" value="1"/>
                           </catDesc>
                        </category>
                     </category>
                     <category xml:id="analysis">
                        <category xml:id="analysis_tools">
                           <catDesc>
                              <ref target="#K5.5">cf. Catalogue 5.5</ref>
                           </catDesc>
                           <catDesc>Does the text collection integrate tools for analyses of the
                              data?</catDesc>
                           <catDesc>
                              <num type="boolean" value="1"/>
                           </catDesc>
                        </category>
                        <category xml:id="analysis_customization">
                           <catDesc>
                              <ref target="#K5.5">cf. Catalogue 5.5</ref>
                           </catDesc>
                           <catDesc>Can the user alter the interface in order to affect the outcomes
                              of representation and analysis of the text collection (besides basic
                              search functionalities), e.g. by applying his or her own queries or by
                              choosing analysis parameters?</catDesc>
                           <catDesc>
                              <num type="boolean" value="1"/>
                           </catDesc>
                        </category>
                     </category>
                     <category xml:id="visualization">
                        <catDesc>
                           <ref target="#K5.6">cf. Catalogue 5.6</ref>
                        </catDesc>
                        <catDesc>Does the text collection provide particular visualizations of the
                           data?</catDesc>
                        <category xml:id="networks">
                           <catDesc>
                              <num type="boolean" value="0"/>
                           </catDesc>
                        </category>
                        <category xml:id="charts">
                           <catDesc>
                              <num type="boolean" value="1"/>
                           </catDesc>
                        </category>
                        <category xml:id="treemaps">
                           <catDesc>
                              <num type="boolean" value="0"/>
                           </catDesc>
                        </category>
                        <category xml:id="wordclouds">
                           <catDesc>
                              <num type="boolean" value="0"/>
                           </catDesc>
                        </category>
                        <category corresp="#other">
                           <catDesc>
                              <num type="boolean" value="0"/>
                           </catDesc>
                           <category corresp="#free">
                              <desc/>
                           </category>
                        </category>
                        <category xml:id="no_visualization">
                           <catDesc>
                              <num type="boolean" value="0"/>
                           </catDesc>
                        </category>
                     </category>
                     <category xml:id="personalization">
                        <catDesc>
                           <ref target="#K5.7">cf. Catalogue 5.7</ref>
                        </catDesc>
                        <catDesc>Is there a personalisation mode that enables the users e.g. to
                           create their own sub-collections of the existing text
                           collection?</catDesc>
                        <catDesc>
                           <num type="boolean" value="1"/>
                        </catDesc>
                     </category>
                  </category>
               </category>
               <category xml:id="preservation">
                  <category xml:id="documentation_project">
                     <catDesc>
                        <ref target="#K6.1">cf. Catalogue 6.1</ref>
                     </catDesc>
                     <catDesc>Does the text collection provide sufficient documentation about the
                        project in general as well as about the aims, contents and methods of the
                        text collection?</catDesc>
                     <catDesc>
                        <num type="boolean" value="0"/>
                     </catDesc>
                  </category>
                  <category xml:id="open_access">
                     <catDesc>
                        <ref target="#K6.2">cf. Catalogue 6.2</ref>
                     </catDesc>
                     <catDesc>Is the text collection Open Access?</catDesc>
                     <catDesc>
                        <num type="boolean" value="0"/>
                     </catDesc>
                  </category>
                  <category xml:id="rights">
                     <category xml:id="rights_declared">
                        <catDesc>
                           <ref target="#K6.2">cf. Catalogue 6.2</ref>
                        </catDesc>
                        <catDesc>Are the rights to (re)use the content declared?</catDesc>
                        <catDesc>
                           <num type="boolean" value="1"/>
                        </catDesc>
                     </category>
                     <category xml:id="rights_license">
                        <catDesc>
                           <ref target="#K6.2">cf. Catalogue 6.2</ref>
                        </catDesc>
                        <catDesc>Under what license are the contents released?</catDesc>
                        <category xml:id="CC0">
                           <catDesc>
                              <num type="boolean" value="0"/>
                           </catDesc>
                        </category>
                        <category xml:id="CC-BY_only">
                           <catDesc>
                              <num type="boolean" value="0"/>
                           </catDesc>
                        </category>
                        <category xml:id="CC-BY-ND">
                           <catDesc>
                              <num type="boolean" value="0"/>
                           </catDesc>
                        </category>
                        <category xml:id="CC-BY-NC">
                           <catDesc>
                              <num type="boolean" value="0"/>
                           </catDesc>
                        </category>
                        <category xml:id="CC-BY-SA">
                           <catDesc>
                              <num type="boolean" value="0"/>
                           </catDesc>
                        </category>
                        <category xml:id="CC-BY-NC-ND">
                           <catDesc>
                              <num type="boolean" value="0"/>
                           </catDesc>
                        </category>
                        <category xml:id="CC-BY-NC-SA">
                           <catDesc>
                              <num type="boolean" value="0"/>
                           </catDesc>
                        </category>
                        <category xml:id="PDM">
                           <catDesc>
                              <num type="boolean" value="0"/>
                           </catDesc>
                        </category>
                        <category corresp="#other">
                           <catDesc>
                              <num type="boolean" value="1"/>
                           </catDesc>
                           <category corresp="#free">
                              <desc/>
                           </category>
                        </category>
                        <category xml:id="no_license">
                           <catDesc>
                              <num type="boolean" value="0"/>
                           </catDesc>
                        </category>
                     </category>
                  </category>
                  <category xml:id="persistent_identification">
                     <catDesc>
                        <ref target="#K6.3">cf. Catalogue 6.3</ref>
                     </catDesc>
                     <catDesc>Are there persistent identifiers and an addressing system for the text
                        collection and/or parts/objects of it and which mechanism is used to that
                        end?</catDesc>
                     <category xml:id="DOI">
                        <catDesc>
                           <num type="boolean" value="0"/>
                        </catDesc>
                     </category>
                     <category xml:id="ARK">
                        <catDesc>
                           <num type="boolean" value="0"/>
                        </catDesc>
                     </category>
                     <category xml:id="URN">
                        <catDesc>
                           <num type="boolean" value="0"/>
                        </catDesc>
                     </category>
                     <category xml:id="PURL.ORG">
                        <catDesc>
                           <num type="boolean" value="0"/>
                        </catDesc>
                     </category>
                     <category corresp="#other">
                        <catDesc>
                           <num type="boolean" value="0"/>
                        </catDesc>
                        <category corresp="#free">
                           <desc/>
                        </category>
                     </category>
                     <category xml:id="persistent_URLs">
                        <catDesc>
                           <num type="boolean" value="1"/>
                        </catDesc>
                     </category>
                     <category corresp="#none">
                        <catDesc>
                           <num type="boolean" value="0"/>
                        </catDesc>
                     </category>
                  </category>
                  <category xml:id="citation">
                     <catDesc>
                        <ref target="#K6.3">cf. Catalogue 6.3</ref>
                     </catDesc>
                     <catDesc>Does the text collection supply citation guidelines?</catDesc>
                     <catDesc>
                        <num type="boolean" value="1"/>
                     </catDesc>
                  </category>
                  <category xml:id="archiving">
                     <catDesc>
                        <ref target="#K6.4">cf. Catalogue 6.4</ref>
                     </catDesc>
                     <catDesc>Does the documentation include information about the long term
                        sustainability of the basic data (archiving of the data)?</catDesc>
                     <catDesc>
                        <num type="boolean" value="0"/>
                     </catDesc>
                  </category>
                  <category xml:id="curation">
                     <catDesc>
                        <ref target="#K6.4">cf. Catalogue 6.4</ref>
                     </catDesc>
                     <catDesc>Does the project provide information about institutional support for
                        the curation and sustainability of the project?</catDesc>
                     <catDesc>
                        <num type="boolean" value="0"/>
                     </catDesc>
                  </category>
                  <category xml:id="completion">
                     <catDesc>
                        <ref target="#K6.4">cf. Catalogue 6.4</ref>
                     </catDesc>
                     <catDesc>Is the text collection completed?</catDesc>
                     <catDesc>
                        <num type="boolean" value="1"/>
                     </catDesc>
                  </category>
               </category>
            </taxonomy>
         </classDecl>
      </encodingDesc>
      <profileDesc>
         <langUsage>
            <language ident="de"/>
         </langUsage>
         <textClass>
            <keywords xml:lang="en">
               <term>20th century</term>
               <term>interface</term>
               <term>italian</term>
               <term>linguistic search</term>
               <term>newspaper</term>
               <term>pos-tagging</term>
               <term>tei</term>
               <term>text collection</term>
            </keywords>
         </textClass>
      </profileDesc>
   </teiHeader>
   <text>
      <front>
         <div type="abstract">
            <p>This paper reviews a huge resource of contemporary Italian newspaper language, the
               «La Repubblica» corpus. The corpus contains articles, which appeared in the Italian
               daily newspaper <emph>La Repubblica</emph> during the years 1985 to 2000 and counts
               more than 380 million tokens. Apart from being tokenized, it is also PoS-tagged,
               enriched with TEI-conformant structural mark-up as well as categorized with respect
               to topics and genres. The data and their preparation are addressed in the first part
               of this paper while its second part deals with access to the corpus. When the review
               was written, there were two possible ways of accessing the corpus: either by the
               ‘old’ interface directly hosted by the Institute of Translational Studies at the
               University of Bologna (SSLMIT) or by the ‘new’ one hosted by a NoSketch Engine. Both
               ways are compared in order to point out the changes.</p>
         </div>
      </front>
      <body>
         <div xml:id="div1">
            <head>Besprechung</head>
            <p xml:id="p1">Aus dem Bedürfnis nach authentischen italienischen Sprachdaten für
               ÜbersetzerInnen ist in den Jahren von 2001 bis 2004 das «La Repubblica»-Korpus am
               Institut für Translatologie der Universität Bologna (Scuola Superiore di Lingue
               Moderne per Interpreti e Traduttori, kurz: SSLMIT) entstanden, so beschreiben es Guy
               Aston und Lorenzo Piccioni. Als das «La Repubblica»-Korpus erstellt wurde, verfügten
               andere Sprachen bereits über große Referenzkorpora, während es im Italienischen
               nichts Vergleichbares gab (cf. Aston / Piccioni 2003). Um diese Lücke zu schließen,
               wurde im Jahr 2001 mit der Erstellung des «La Repubblica»-Korpus begonnen, das in
               dieser Rezension beleuchtet werden soll.</p>
            <p xml:id="p2">Mittlerweile existieren zwar einige Korpora italienischer
                  Gegenwartssprache,<note xml:id="ftn1">Neben dem Korpus CORIS/CODIS (Rossini
                  Favretti, Rema et al. 1998-2017) beispielsweise das Perugia Corpus CEP (cf. Spina
                  2014) oder das Projekt CLIPS (cf. Sobrero 2007).</note> seine Größe und
               Aufbereitung sowie der ausgedehnte Erhebungszeitraum lassen das «La
               Repubblica»-Korpus aber auch knapp 13 Jahre nach seiner Fertigstellung noch attraktiv
               für verschiedene Nutzungsinteressen (s.u.) erscheinen. Im Folgenden sollen zunächst
               (1) Umfang, (2) Daten, (3) Markup und (4) inhaltliche bzw. linguistische Annotation
               des Korpus beschrieben werden, ehe auf dessen Nutzungsmöglichkeiten durch
               verschiedene Interfaces eingegangen wird.</p>
            <p xml:id="p3">
               <emph>Umfang.</emph> Mit insgesamt 380 Millionen Tokens ist das «La
               Repubblica»-Korpus knapp dreimal so groß wie das <emph>British National Corpus</emph>
               (Bodleian Libraries (University of Oxford) on behalf of the BNC Consortium 2007) mit
               ca. 100 Millionen Tokens oder das italienische Referenzkorpus CORIS/CODIS (Rossini
               Favretti, Rema et al. 1998-2017), das ca. 130 Millionen Tokens umfasst und noch in
               den Kinderschuhen steckte, als das «La Repubblica»-Korpus konzipiert wurde (cf. Aston
               / Piccioni 2007).</p>
            <p xml:id="p4">
               <emph>Daten.</emph> Das Korpus basiert auf Texten, die im Zeitraum von 1985 bis 2000
               in der italienischen Tageszeitung «La Repubblica» erschienen sind, einer der meist
               gelesenen Tageszeitungen Italiens (cf. Baroni, Bernardi, Comastri et al. 2004:
                  1771).<note xml:id="ftn2">Wie zum Zeitpunkt der Korpuserstellung (cf. Baroni /
                  Bernardi / Comastri et al. 2004: 1771) rangiert <emph>La Repubblica</emph> auch
                  aktuell knapp hinter dem <emph>Corriere della Sera</emph> auf Platz 2 der
                  meistgelesenen allgemeinen Tageszeitungen (diese Aussage basiert auf Daten von
                  Audipress (2017, letzter Zugriff: 10.07.2017).</note> Allerdings waren es keine
               gedruckten Zeitungsausgaben, die als unmittelbare Datengrundlage fungierten, sondern
               16 CD-ROMs, die von «La Repubblica» erstellt und vertrieben wurden. Jede CD-ROM
               enthielt Aston und Piccioni (2003) zufolge eine Datenbank mit den Artikeln und
               zugehörigen Metadaten der «La Repubblica»-Ausgaben eines gesamten Jahres. Bilder und
               Tabellen bzw. andere Verzeichnisse waren in der Datenbank ebenso wenig präsent wie
               Werbung oder Beilagen (cf. Aston / Piccioni 2003). Bei der Erstellung des Korpus
               wurden die Texte und verfügbaren Metadaten zunächst von den CD-ROMs extrahiert und
               als ASCII-Dateien gespeichert. Bei der Extraktion musste berücksichtigt werden, dass
               die Kodierung der Daten bei manchen CD-ROMs variierte (cf. Aston / Piccioni 2003).
               Nach der Extraktion wurden die Texte zunächst normalisiert, um das Korpus mittels
                  <emph>Corpus Workbench</emph>
               <note xml:id="ftn3">Noah Bubenhofer (2006-2015)
                  beschreibt die <emph>Corpus Workbench</emph> als „Konkordanz- und
                  Korpusanalyse-Software, mit der eigene Korpora, die mit linguistischen
                  Annotationen versehen sind, bearbeitet werden können“.</note> zu indizieren.
               Normalisiert wurden nach Angaben von Aston und Piccioni (2003) u.a. Akzente („perché“
               versus „perchè“ versus „perche'“), wobei die Variante mit nachgestelltem Apostroph
               zugunsten einer am Standarditalienischen orientierten Akzentuierung im Wort
               („perché“) aufgegeben wurde. Weiterhin wurden Sonderzeichen (z.B. Apostroph,
               Anführungszeichen) durch Entitätsreferenzen (z.B. &amp;apos für den Apostroph)
               ersetzt.</p>
            <p xml:id="p5">
               <figure xml:id="img1">
                  <graphic url="http://ride.i-d-e.de/wp-content/uploads/issue_6/larepubblica/pictures/picture-1.png"/>
                  <head type="legend">Struktur eines XML-Dokumentes des <emph>«La
                     Repubblica»</emph>-Korpus (aus: Aston / Piccioni 06.05.2003).</head>
               </figure>
               <emph>Markup.</emph> Die Artikel und ihre Metadaten wurden im Anschluss an die
               Normalisierung in XML-Dokumenten neu strukturiert und zwar unter Berücksichtigung der
                  TEI-Richtlinien.<note xml:id="ftn4">Gemeint sind die <emph>TEI Guidelines for
                     Electronic Text Encoding and Interchange</emph> (TEI Consortium 04.10.2015),
                  die mittlerweile als Version P5 verfügbar sind.</note> Aston und Piccioni (2003)
               haben die Strukturierung der Daten in einer Grafik veranschaulicht (vgl. <ref type="crossref" target="#img1">Abb. 1</ref>). Pro Zeitungsausgabe wurde ein
               TEI-Dokument erstellt, das jeweils aus Header (<code>&lt;teiHeader&gt;</code>) und
               Textkörper (<code>&lt;text&gt;</code>) besteht. Letzterer wurde weiter untergliedert
               in <code>&lt;div&gt;</code>-Elemente, die die einzelnen Artikel einer Zeitungsausgabe
               beinhalten. Die Artikelstruktur gliedert sich wiederum in verschiedene, teilweise
               fakultative Elemente: Überschriften (<code>&lt;head&gt;</code>), Byline
                  (<code>&lt;byline&gt;</code>) und Absatz (<code>&lt;p&gt;</code>), wobei der
               Absatz noch einmal in Sätze unterteilt wurde. Die Entscheidung, den Textkörper eines
               jeden Artikels in nur einem Absatz wiederzugeben, war Aston und Piccioni (2003)
               zufolge der Beschaffenheit der Rohdaten geschuldet, denn den Datenbanken waren (außer
               den Überschriften) keine Informationen über die Strukturierung eines Artikels zu
               entnehmen. Die Untergliederung eines Artikels in Sätze (<code>&lt;s&gt;</code>)
               konnte ebenfalls nicht unmittelbar aus den Rohdaten extrahiert werden, sondern
               bedurfte einer nachträglichen Segmentierung (cf. Aston / Piccioni 2003). Verfügbare
               Metadaten zu den einzelnen Artikeln wurden in den Attributen der
                  <code>&lt;div&gt;</code>-Tags festgehalten, etwa die zugehörige Seitenzahl, die
               Textsorte oder das Thema des Artikels.<note xml:id="ftn5">Aston und Piccioni
                  (06.05.2003) erwähnen nicht alle oben genannten Attribute, aber die übrigen,
                  article.top und article.gen, lassen sich anhand der Korpus-Informationen auf dem
                     <emph>NoSketch Engine</emph> Interface (CoLiTec: o.J.) erschließen.</note> Nach
               Bernardi, Baroni, Comastri und anderen (2004: 1771) besitzen nicht nur die jeweiligen
               Dokumente, in denen die Daten einer Zeitungsausgabe eingespeist wurden, einen Header,
               sondern das Korpus insgesamt. Der Header für das gesamte Korpus enthält dabei
               Metainformationen, die das Korpus als Ganzes betreffen, etwa editorische Prinzipien
               oder Urheberrechtsbeschränkungen. Für Korpus-NutzerInnen einsehbar ist der Header
               nach meinem bisherigen Erkenntnisstand (12.07.2017) nicht.</p>
            <p xml:id="p6">
               <emph>Inhaltliche bzw. linguistische Annotation.</emph> Das Korpus wurde aber nicht
               nur mit strukturellen Metadaten angereichert, sondern auch mit inhaltlichen und
               linguistischen Annotationen. Es ist mit PoS-Tags annotiert (cf. Baroni / Bernardi /
               Comastri et al. 2004: 1772), wobei eine reduzierte und modifizierte Form des
                  EAGLES-Tagsets<note xml:id="ftn6">Das Akronym EAGLES steht für <emph>Expert
                     Advisory Group on Language Engineering Standards</emph>, eine Initiative der
                  Europäischen Kommission, die zum Ziel hatte, De-facto-Standards unter anderem für
                  die Erstellung, Beschreibung und Repräsentation von großen Sprachressourcen zu
                  etablieren (cf. EAGLES 24.06.2014).</note> verwendet wurde. Das Tagset wurde
               zunächst auf ein Subkorpus bestehend aus 180 zufällig ausgewählten Artikeln
               angewendet. Dieses automatisch annotierte und manuell kontrollierte Subkorpus
               erfüllte zwei Funktionen: zum einen wurde es verwendet, um zu eruieren, welche Tagger
               bzw. Kombinationen von Taggern die verlässlichsten Ergebnisse produzieren würden und
               diente sozusagen als Kontrollkorpus; zum anderen diente es als Trainingskorpus für
               diejenige Kombination von Taggern, die sich mit einer durchschnittlichen
                  Genauigkeit<note xml:id="ftn7">Hier ist das arithmetische Mittel gemeint. Angaben
                  zur Streuung wurden nicht gemacht.</note> von 95,46 % als verlässlichste erwiesen
               hat. Mit dieser verlässlichsten Kombination wurde dann der Rest des Korpus
               automatisch annotiert. Für KorpusnutzerInnen ist nicht ersichtlich, welche Teile des
               Korpus manuell bearbeitet wurden und welche ausschließlich durch automatisierte
               Prozesse. Im Zuge einer weiteren inhaltlichen / linguistischen Annotation wurden die
               Zeitungsartikel nach Texttyp (Genre) und Thema kategorisiert (cf. Baroni / Bernardi /
               Comastri et al. 2004: 1772). Auch dies geschah zunächst manuell anhand von 15.000
               Artikeln, ehe eine Support Vector Machine (<emph>SVMLight</emph>), mit einer
               Genauigkeit von 90,3% (cf. Baroni / Bernardi / Comastri et al. 2004: 1773) auf den
               Rest des Korpus angewendet wurde. Es werden nach Angaben der AutorInnen insgesamt
               zwei Texttypen (<emph>news-report</emph> und <emph>comment</emph>) sowie 10 Themen
               (u.a. <emph>culture</emph> und <emph>economics</emph>) unterschieden.<note xml:id="ftn8">Anhand der Korpus-Informationen auf dem <emph>NoSketch Engine</emph>
                  Interface (CoLiTec: o.J.) ist ersichtlich, dass die Themen in der Zwischenzeit
                  noch weiter differenziert wurden, Politik etwa u.a. in Schulpolitik und
                  Sportpolitik.</note> Dass das Korpus auch lemmatisiert wurde, ist u.a. der
               Online-Beschreibung des Korpus (cf. DIT 24.05.2017) zu entnehmen. Daraus geht hervor,
               dass die einzelnen Tokens mit Hilfe von <emph>Morph-It!</emph>
               <note xml:id="ftn9">
                  <emph>Morph-It!</emph> ist ein frei nutzbares „lexicon of inflected forms with
                  their lemma and morphological features“ (cf. SSLMIT Dev Online 2009), das
                  ebenfalls an der Universität Bologna, und zwar von Marco Baroni und Eros Zanchetta
                  entwickelt wurde und Lemmatizern als Datengrundlage dienen kann.</note> annotiert
               wurden.</p>
            <p xml:id="p7">Ein solches PoS-annotiertes, lemmatisiertes und in Bezug auf Texttyp und
               Thema kategorisiertes Zeitungskorpus (cf. SSLMIT Dev Online 2004b) kann vielfältige
               Erkenntnisinteressen befriedigen, auch wenn es auf nur einer Quelle, «La Repubblica»,
               basiert. Guy Astone und Lorenzo Piccione (2003) nennen neben der Nutzung in
               didaktischen Kontexten auch diachrone und synchrone Forschungsmöglichkeiten und
               insbesondere die Untersuchung kontrastiver Fragestellungen. Bereits genutzt wurde das
               Korpus beispielsweise von Lorenza Pescia für die Untersuchung sexistischen
               Sprachgebrauchs in der Presse Italiens und der Schweiz (cf. Pescia 2010). Vorstellbar
               wären auch Untersuchungen zum <emph>Italiano neo-standard</emph>
               <note xml:id="ftn10">Der Begriff "neo-standard" geht auf Gaetano Berruto (1987: 62) zurück und
                  bezeichnet mit den Worten Eduardo Blasco-Ferrers (1994: 217) eine "Ist-Norm" des
                  Italienischen, die sich u.a. durch den Sprachgebrauch in verschiedenen Arten von
                  Massenmedien entwickelt hat und sich immer stärker im öffentlichen Sprachgebrauch
                  manifestiert. Sie steht in Kontrast zu einer "Soll-Norm", auf die sich etwa
                  Italienisch-Lehrwerke oder präskriptive Grammatiken berufen (haben).</note>, da
               Zeitungen nach Ansicht von Massimo Cerruti, Claudia Crocco und Stefania Marzo (2017:
               9) besonders empfänglich für diese Variante des Italienischen sind. Um derartige
               Untersuchungen durchführen zu können, bedarf es des – idealiter freien – Zugangs zu
               Zeitungskorpora des Italienischen.</p>
            <p xml:id="p8">Das «La Repubblica»-Korpus ist zumindest frei durchsuchbar, kann aber auf
               Grund von Urheberrechts- und Nutzungsvereinbarungen mit «La Repubblica» nicht
               gänzlich eingesehen werden (cf. Baroni / Bernardini / Comastri et al. 2004: 1773). Um
               es zu durchsuchen bieten sich momentan (Stand: 11.07.2017) sogar zwei Möglichkeiten,
               denn die Rezension ist zu einem Zeitpunkt entstanden, als das Korpus im Begriff war,
               von einem Interface auf ein anderes umzuziehen, nämlich (A) vom Portal SSLMIT Online
               (2004a) auf (B) das Dipintra-Portal (CoLiTec: o.J.), das auf <emph>NoSketch
                  Engine</emph>
               <note xml:id="ftn11">
                  <emph>NoSketch Engine</emph> ist die „open source version of Sketch Engine with
                  certain functionality limitations“ (Lexical Computing CZ s.r.o. 2017a). Sketch
                  Engine wiederum ist eine Software zum Korpusmanagement und zur
                  Korpusabfrage.</note> basiert. Gerade für NutzerInnen der bisherigen Oberfläche A
               kann ein Vergleich der beiden Zugriffsmöglichkeiten hilfreich sein, um zu erfahren,
               welche Funktionen auf der neuen Oberfläche B wie umgesetzt wurden, welche Funktionen
               entfallen und welche neu hinzukommen. Zudem wird durch einen Vergleich ersichtlich,
               welche Vorteile der Umzug auf ein neues Portal in diesem Fall mit sich bringt.
               Deshalb sollen die beiden Oberflächen (kurz: A und B) einander unter Berücksichtigung
               folgender Punkte gegenübergestellt werden: (1) Nutzungsbedingungen, (2)
               Dokumentation, (3) Suchoptionen, (4) Ergebnisanzeige, (5) Ergebnisverarbeitung und
               (6) Hilfestellungen.</p>
            <p xml:id="p9">
               <emph>Nutzungsbedingungen.</emph> Für die Nutzung von A ist eine Registrierung
               erforderlich, die zügig und automatisiert funktioniert. Das per E-Mail zugesandte
               Passwort lässt sich allerdings nicht ändern. B kann ohne vorherige Anmeldung genutzt
               werden.</p>
            <p xml:id="p10">
               <emph>Dokumentation.</emph> Beide Oberflächen informieren mittels Überblicksseiten
               über das Korpus (Umfang, Daten, Markup, Annotationen) und verweisen auf ein Dokument
               mit veralteten weiterführenden Informationen, nämlich den Artikel von Baroni, Silvia
               Bernardini, Federica Comastri und anderen (2004). Tatsächlich findet sich keine
               aktuellere Beschreibung des Korpus, obwohl es sich in der Zwischenzeit, etwa
               hinsichtlich der Zuordnung von Themen zu Artikeln (vgl. <ref type="crossref" target="#ftn8">Fußnote 8</ref>), verändert hat. Informationen über die
               Normalisierung der Daten finden sich verteilt auf verschiedene Artikel (Baroni /
               Bernardini / Comastri et al. 2004; Aston / Piccioni 2003) und sind nicht unmittelbar
               auf den Oberflächen einsehbar. Sie sind allerdings nicht unwichtig für manche
               Forschungsinteressen: Wie Aston und Piccioni (06.05.2003) anmerken, kann
               beispielsweise die Richtung des Wortakzentes („perché“ versus „perchè“) auf
               sozio-geografische Variation zurückgeführt werden. Durch eine Normalisierung bzw.
               Standardisierung der Akzentsetzung sind Untersuchungen einer solchen Variation mit
               dem «La Repubblica»-Korpus nicht möglich.</p>
            <p xml:id="p11">Eine weitere an keiner Stelle (Stand: 11.07.2017) dokumentierte
               Veränderung ist die Einbindung von Subkorpora in die Oberfläche B. Die Suche kann
               dort beschränkt werden auf Subkorpora wie „GelareRepubblica“, „VER:fin“ oder „c“,
               deren Komposition nicht selbsterklärend ist und über die bisher nur bekannt gegeben
               wird, wie viele Tokens sie umfassen.</p>
            <p xml:id="p12">
               <figure xml:id="img2">
                  <graphic url="http://ride.i-d-e.de/wp-content/uploads/issue_6/larepubblica/pictures/picture-2.png"/>
                  <head type="legend">Strukturelle Elemente und Attribute Interface A.</head>
               </figure>
               <figure xml:id="img3">
                  <graphic url="http://ride.i-d-e.de/wp-content/uploads/issue_6/larepubblica/pictures/picture-3.png"/>
                  <head type="legend">Strukturelle Elemente und Attribute Interface B.</head>
               </figure> Welche XML-Elemente, Attribute und Werte beim strukturellen Markup
               verwendet wurden, ist bei A unmittelbar ersichtlich (vgl. <ref type="crossref" target="#img2">Abb. 2</ref>), während die Attribute und deren Werte bei Interface
               B erst sichtbar werden, wenn auf die zugehörigen Elemente bzw. Attribute geklickt
               wird (vgl. <ref type="crossref" target="#img3">Abb. 3</ref>, nach Klick auf das
               Element <code>&lt;article&gt;</code>). Den Hinweis, dass zusätzliche Klicks
               erforderlich sind, bekommen NutzerInnen erst, wenn sie mit dem Cursor über die
               Elemente fahren. Welche PoS-Tags verwendet wurden, lässt sich dagegen auf den
               Übersichtsseiten beider Oberflächen klarer erschließen: A verweist auf einen Link zum
               Tagset, während B das Tagset in der Korpus-Information (cf. DIT 24.05.2017)
               unmittelbar anzeigt. Einblick in den TEI-Header des gesamten Korpus könnte zu einer
               zuverlässigeren und aktuelleren Dokumentation beitragen.</p>
            <p xml:id="p13">
               <figure xml:id="img4">
                  <graphic url="http://ride.i-d-e.de/wp-content/uploads/issue_6/larepubblica/pictures/picture-4.png"/>
                  <head type="legend">Abfragemaske Interface A.</head>
               </figure>
               <figure xml:id="img5">
                  <graphic url="http://ride.i-d-e.de/wp-content/uploads/issue_6/larepubblica/pictures/picture-5.png"/>
                  <head type="legend">Abfragemaske Interface B.</head>
               </figure>
               <emph>Suchoptionen.</emph> Beide Oberflächen bieten zwei Arten von Korpusabfragen an,
               nämlich die Erstellung von Wortlisten und die Ausgabe von Konkordanzen. Während
               Wortlisten bei A nur über CQP-Abfragen generiert werden können (vgl. <ref type="crossref" target="#img4">Abb. 4</ref>), besitzt B eine Abfragemaske mit
               verschiedenen Optionen (vgl. <ref type="crossref" target="#img5">Abb. 5</ref>). Um
               bei A Wortlisten zu erstellen, müssen sich NutzerInnen mit der Syntax von CQP, der
               Abfragesprache der Korpus-Workbench, auseinandersetzen. Um etwa nach allen Wörtern zu
               suchen, die mit „donn“ beginnen, muss neben regulären Ausdrücken (. oder *) bekannt
               sein, wie nach Wörtern statt nach Lemmata gesucht wird und wie die Suche auf einzelne
               Wörter (word+0) im Gegensatz zu Bi- oder N-Grammen (word+1) beschränkt wird. Für die
               gleiche Suche auf Interface B muss dagegen keine Abfragesprache beherrscht werden.
               Neben der Kenntnis regulärer Ausdrücke ist es hier allerdings besonders wichtig, die
               vielfältigen Such- und Ausgabeoptionen zu verstehen.</p>
            <p xml:id="p14">Statt auf das gesamte Korpus kann die Suche bei B auch nur auf
               Subkorpora bezogen werden. Zudem kann nicht nur nach Wortformen, sondern auch nach
               Lemmata, Tags und sogar nach Werten zu oben genannten Attributen gesucht werden, was
               bei A zwar ebenfalls möglich ist, aber erst durch komplexere CQP-Formulierungen.
               Filtermöglichkeiten bei B bestehen zum einen bezüglich der Häufigkeit, mit der
               Suchobjekte im Korpus auftreten sollen, um in der Wortliste berücksichtigt zu werden;
               zum anderen können anhand von White- / und Blacklists Listen bestimmter Suchobjekte
               berücksichtigt werden, die bei der Suche ein- oder ausgeschlossen werden sollen. Ein
               Beispiel für eine Blacklist zur oben genannten Suchabfrage ist ein Textdokument, das
               die Wortformen „donna“ und „donne“ enthält. Diese beiden Wortformen würden in der
               Wortliste nicht berücksichtigt werden.</p>
            <p xml:id="p15">Für die Konkordanzabfrage bieten beide Oberflächen sowohl einen
               einfachen (simple) als auch einen oder mehrere fortgeschrittene Suchtypen an. Bei A
               bedeutet ‚simple‘, dass für die Suche eine Abfragemaske verwendet wird, während die
               fortgeschrittene Suche vollständig auf der Verwendung von CQP basiert. Bei Interface
               B sind alle Suchtypen mit einer Suchmaske verknüpft, die verschiedene Filter- und
               Ausgabeoptionen bereithält. Dort bedeutet ‚simple‘, dass anhand der Eingabe versucht
               wird, festzustellen, ob die Suche auf ein Lemma, eine Wortform oder eine Phrase
               abzielt (cf. Lexical Computing CZ s.r.o. 2017b). Alternativen zur einfachen Suche
               sind bei B die gezielte Suche nach Lemmata, Phrasen, Wortformen, Zeichen oder eine
               Suche mit Hilfe der Abfragesprache CQL.</p>
            <p xml:id="p16">
               <figure xml:id="img6">
                  <graphic url="http://ride.i-d-e.de/wp-content/uploads/issue_6/larepubblica/pictures/picture-6.png"/>
                  <head type="legend">Vergleich der Suchoptionen auf Interface A und B.</head>
               </figure>Da die Suchmasken beider Oberflächen zu viele Optionen besitzen, um auf jede
               gesondert einzugehen, werden sie tabellarisch gegenübergestellt und nur ausgewählte
               Optionen ausführlicher besprochen. Bezüglich aller grau unterlegten Suchoptionen
               (vgl. <ref type="crossref" target="#img6">Abb. 6</ref>), haben sich beim Umzug des
               Korpus von Interface A auf Interface B Änderungen ergeben. Da über die Hälfte der
               Tabelle grau unterlegt ist, kann das Ausmaß der Veränderungen erahnt werden.</p>
            <p xml:id="p17">Die Einschränkungsmöglichkeit der Suche auf bestimmte Genres (Kommentar
               / News) oder Jahre ist gleich geblieben. Bei B kann die Suche nicht nur auf Themen,
               sondern auch auf Subthemen beschränkt werden. B bietet zusätzlich die Möglichkeit,
               die Suche auf ein Subkorpus oder / und auf Artikel mit einer bestimmten ID, von
               bestimmten AutorInnen, mit einer bestimmten Länge oder mit einem bestimmten Titel zu
               beschränken. Da die Benennungsprinzipien der IDs nicht selbsterklärend sind und auch
               nicht erklärt werden, stellt sich allerdings die Frage, wie die Suche nach Artikeln
               mit einer speziellen ID sinnvoll eingesetzt werden kann. Im Gegensatz zu B ermöglicht
               A, Diakritika zu berücksichtigen bzw. zu ignorieren. Zudem ist es bei B nicht mehr
               möglich, die Trefferzahl im Vorfeld zu begrenzen. Nach Durchführung einer Suchanfrage
               ist es aber weiterhin möglich, andere Anzeigeoptionen (u.a. Treffer pro Seite /
               Sortierung) einzustellen.</p>
            <p xml:id="p18">
               <figure xml:id="img7">
                  <graphic url="http://ride.i-d-e.de/wp-content/uploads/issue_6/larepubblica/pictures/picture-7.png"/>
                  <head type="legend">Optionen der Ergebnisanzeige bei Interface B.</head>
               </figure>
               <emph>Ergebnisanzeige.</emph> Die Ergebnisanzeige bei B umfasst im Vergleich zu A
               wesentlich mehr Möglichkeiten (vgl. <ref type="crossref" target="#img7">Abb.
               7</ref>). Grundsätzlich wird neben der KWIC-Ausgabe auch eine „Sentence“-Ausgabe
               angeboten, bei der jedes Keyword eingebettet in seinen zugehörigen Satz angezeigt
               wird. Neu bei Oberfläche B ist zudem, dass entschieden werden kann, ob zusätzliche
               Informationen (z.B. PoS-Tag oder Lemma) nur für das jeweilige Keyword angezeigt
               werden sollen, oder auch für die Tokens, die das Keyword umgeben. Eine weitere
               Besonderheit bei B ist die Sortierung nach GDEX (Good Dictionary EXamples),<note xml:id="ftn12">Das automatische Erkennen von GDEX ist eine Besonderheit von Sketch
                  Engine. Für weitere Informationen bezüglich GEDX siehe Kilgarriff, Husák, Mc Adam
                  und andere (2008).</note> die besonders für eine lexikographische Nutzung
               interessant ist.</p>
            <p xml:id="p19">Während sich die Informationen zu den einzelnen Treffern bei A in der
               Angabe der Korpusposition erschöpfen, können bei B nun Metadaten zu den zugehörigen
               Artikeln und ein erweiterter sprachlicher Kontext angezeigt werden. Eine
               randomisierte Darstellung der Suchergebnisse kann bei A bereits vor der Suche
               angefordert werden. Dies ist bei B erst nach Durchführung der Suche möglich und
               gehört damit bereits zum nächsten Punkt, der Ergebnisverarbeitung.</p>
            <p xml:id="p20">
               <emph>Ergebnisverarbeitung.</emph> Bei A können lediglich Suchhistorien gespeichert
               und Wortlisten als Textdateien per E-Mail zugesandt werden. B bietet erneut deutlich
               mehr Möglichkeiten der Ergebnisverarbeitung und –sicherung: Die Treffer können nach
               verschiedenen Kriterien sortiert und gefiltert werden. Auch können Samples aus
               (reproduzierbar) zufällig ausgewählten Treffern erstellt werden. Sogar ganze
               Subkorpora sind erstell- und als XML- oder TXT-Datei speicherbar.</p>
            <p xml:id="p21">
               <figure xml:id="img8">
                  <graphic url="http://ride.i-d-e.de/wp-content/uploads/issue_6/larepubblica/pictures/picture-8.png"/>
                  <head type="legend">Häufigkeitsverteilung von „donna“ über 100 gleichgroße Teile
                     des Korpus.</head>
               </figure> Es ist bei B möglich, Keywords und die sie umgebenden Tokens hinsichtlich
               ihrer Auftretenshäufigkeit und Verteilung im Korpus zu analysieren. So kann
               beispielsweise angezeigt werden, in welchen thematischen Kontexten ein Keyword wie
               häufig vorkommt. Aber auch die sprachliche Umgebung der Keywords kann in Form von
               Kollokationen analysiert werden. Schließlich wird sogar eine Form der Visualisierung
               angeboten: Beim Klick auf „Visualize“ öffnet sich ein Balkendiagramm, das die
               Verteilung eines Keywords im Korpus repräsentiert. <ref type="crossref" target="#img8">Abb. 8</ref> zeigt die Verteilung der Wortform „donna“. Das Korpus
               wird dabei in gleichgroße Teile zerlegt, wobei der Feinheitsgrad eingestellt werden
               kann (cf. Lexical Computing CZ s.r.o. 2017b). Beim Klick auf einen Balken werden die
               Treffer aus dem zugehörigen Korpusteil angezeigt. Wie beschrieben bietet B eine
               Vielzahl von Such-, Anzeige- und Verarbeitungsoptionen an, die zu einem großen Teil
               nicht selbsterklärend sind. Deshalb sind Hilfestellungen für eine optimale Nutzung
               der Oberfläche unabdingbar.</p>
            <p xml:id="p22">
               <emph>Hilfestellungen.</emph> Während die Hilfsseiten bei A problemlos gelesen werden
               können, funktionieren die Links bei B derzeit (Stand: 12.07.2017) nicht. Die Seiten
               lassen sich aber über die Homepage von <emph>Sketch Engine</emph> relativ leicht
               selbst finden. Das „User manual“ (cf. Lexical Computing CZ s.r.o 2017c) bietet eine
               verständliche Einführung in die Funktionen und zugrundeliegenden Statistiken von
                  <emph>Sketch Engine</emph>. Das dort erworbene Wissen kann dann auf die
                  <emph>NoSketch Engine</emph>-Oberfläche des «La Repubblica»-Korpus übertragen
               werden.</p>
         </div>
         <div xml:id="div2">
            <head>Fazit</head>
            <p xml:id="p23">Zusammenfassend zeigt der Vergleich, dass Oberfläche B über mehr,
               vielfältigere und komplexere Optionen der Suche, Anzeige und Verarbeitung der Daten
               des «La Repubblica»-Korpus verfügt, die stärker über Eingabemasken angewählt werden
               können und nicht notwendig die Kenntnis einer Abfragesprache erfordern. Um Oberfläche
               B aber voll nutzen zu können, sollten Korpusdokumentation und die Links zu
               entsprechenden Hilfsseiten aktualisiert werden.</p>
            <p xml:id="p24">Wer das «La Repubblica»-Korpus über die neue Oberfläche ansteuert, kann
               sich auch weiterhin auf die Möglichkeit zur Erschließung unentgeltlich nutzbarer
               authentischer italienischer Sprachdaten beachtlichen Ausmaßes einstellen; sei es zu
               Zwecken der automatischen Sprachverarbeitung, für kontrastive Untersuchungen, für
               Übersetzungen, in didaktischen Kontexten oder für linguistische Forschungsinteressen.
               Die Annotation der Rohdaten des Korpus mit linguistischen (PoS-Tags), inhaltlichen
               (Thema und Texttyp) und strukturellen Informationen legen zusammen mit der
               Lemmatisierung den Grundstein für eine Vielzahl unterschiedlichster Korpusabfragen,
               die mittels Weboberfläche(n) realisiert werden können. </p>
         </div>
      </body>
      <back>
         <div type="bibliography">
            <listBibl>
               <bibl>Aston, Guy, and Lorenzo Piccioni. 2003. "Un grande corpus di italiano
                  giornalistico." Guy Aston. Pagina personale/Personal page. Last modified May 6,
                  2003. Accessed April 5, 2017. <ref target="http://www.sslmit.unibo.it/~guy/aitla_repubblica.htm">http://www.sslmit.unibo.it/~guy/aitla_repubblica.htm</ref>.</bibl>
               <bibl>Audipress. 2017. "Scenario Quotidiani." Last modified May 30, 2017. Accessed
                  July 10, 2017. <ref target="http://audipress.it/audipress-sito-2017/wp-content/uploads/2017/05/Dati-Audip-2017_I_invio_completo.xlsx">http://audipress.it/audipress-sito-2017/wp-content/uploads/2017/05/Dati-Audip-2017_I_invio_completo.xlsx</ref>. </bibl>
               <bibl>Baroni, Marco, Silvia Bernardini, Fedrica Comastri, Lorenzo Piccioni,
                  Alessandra Volpi, Guy Aston, and Marco Mazzoleni. 2004. "Introducing the <emph>«La
                     Repubblica»</emph> Corpus: A Large, Annotated, TEI(XML-) Compliant Corpus of
                  Newspaper Italian" In <emph>Proceedings of LREC 2004</emph>, 1771–1774. <ref target="https://web.archive.org/web/20170912102519/http://clic.cimec.unitn.it/marco/publications/lrec2004/rep_lrec_2004.pdf">https://web.archive.org/web/20170912102519/http://clic.cimec.unitn.it/marco/publications/lrec2004/rep_lrec_2004.pdf</ref>. </bibl>
               <bibl>Berruto, Gaetano. 1987. <emph>Sociolinguistica dell'Italiano
                     contemporaneo</emph> (= Studi Superiori NIS 33). Roma: NIS.</bibl>
               <bibl>Blasco Ferrer, Eduardo. 1994. <emph>Handbuch der italienischen
                     Sprachwissenschaft</emph> (= Grundlagen der Romanistik 16). Berlin: Erich
                  Schmidt.</bibl>
               <bibl>Bodleian Libraries (University of Oxford) on behalf of the BNC Consortium.
                  2007. <emph>The British National Corpus, version 3.</emph> Accessed July 10, 2017.
                     <ref target="http://www.natcorp.ox.ac.uk/">http://www.natcorp.ox.ac.uk/</ref>. </bibl>
               <bibl>Bubenhofer, Noah. 2006-2015. "Die Arbeit mit der IMS Open Corpus Workbench am
                  Beispiel des Text+Berg-Korpus." In <emph>Einführung in die Korpuslinguistik.
                     Praktische Grundlagen und Werkzeuge.</emph>
                  <ref target="https://web.archive.org/web/20170912103227/http://www.bubenhofer.com/korpuslinguistik/kurs/index.php?id=cwb_start.html">https://web.archive.org/web/20170912103227/http://www.bubenhofer.com/korpuslinguistik/kurs/index.php?id=cwb_start.html</ref>. </bibl>
               <bibl>Cerruti, Massimo, Claudia Crocco, and Stefania Marzo. 2017. "On the development
                  of a new standard norm in Italian." In <emph>Towards a New Standard. Theoretical
                     and Empirical Studies on the Restandardization of Italian</emph> (= Language
                  and Social Life 6), edited by Cerruti, Massimo, Claudia Crocco, and Stefania
                  Marzo, 3–28. Boston / Berlin: de Gruyter.</bibl>
               <bibl>CoLiTec. N.d. <emph>Corpora Dipintra.</emph> Bologna: Università di Bologna.
                     <ref target="https://web.archive.org/web/20170912103703/https://corpora.dipintra.it/public/run.cgi/first_form">https://web.archive.org/web/20170912103703/https://corpora.dipintra.it/public/run.cgi/first_form</ref>. </bibl>
               <bibl>DIT (Dipartimento Interpretazione e Traduzione). 2017. "Repubblica."
                  Documentation Wiki. Last modified May 24, 2017. Bologna: Università di Bologna.
                     <ref target="https://web.archive.org/web/20170912104442/http://docs.sslmit.unibo.it/doku.php?id=corpora:repubblica">https://web.archive.org/web/20170912104442/http://docs.sslmit.unibo.it/doku.php?id=corpora:repubblica</ref>. </bibl>
               <bibl>EAGLES. 2014. "EAGLES. Expert Advisory Group on Language Engineering
                  Standards." Last modified June 24, 2014. <ref target="https://web.archive.org/web/20170912104707/http://www.echo.lu/langeng/en/lre1/eagles.html">https://web.archive.org/web/20170912104707/http://www.echo.lu/langeng/en/lre1/eagles.html</ref>. </bibl>
               <bibl>Kilgarriff, Adam, Miloš Husák, Katy McAdam, et al. 2008. "GDEX: Automatically
                  finding good dictionary examples in a corpus" In <emph>Proceedings of the 13th
                     EURALEX International Congress</emph>, 425–432. Spain, July 2008.</bibl>
               <bibl>Lexical Computing CZ s.r.o. 2017a. "NoSketch Engine and Sketch Engine. What is
                  the difference?" In <emph>Sketch Engine</emph>, edited by Lexical Computing CZ
                  s.r.o. <ref target="https://web.archive.org/web/20170912105438/https://www.sketchengine.co.uk/nosketch-engine/">https://web.archive.org/web/20170912105438/https://www.sketchengine.co.uk/nosketch-engine/</ref>. </bibl>
               <bibl>Lexical Computing CZ s.r.o. 2017b. "How to generate a concordance?" In
                     <emph>Sketch Engine</emph>, edited by Lexical Computing CZ s.r.o. <ref target="https://web.archive.org/web/20170912105707/https://www.sketchengine.co.uk/user-guide/user-manual/concordance-introduction/concordance-search/">https://web.archive.org/web/20170912105707/https://www.sketchengine.co.uk/user-guide/user-manual/concordance-introduction/concordance-search/</ref>. </bibl>
               <bibl>Lexical Computing CZ s.r.o. 2017c. "User manual." In <emph>Sketch
                  Engine</emph>, edited by Lexical Computing CZ s.r.o. <ref target="https://web.archive.org/web/20170912105844/https://www.sketchengine.co.uk/user-guide/user-manual/">https://web.archive.org/web/20170912105844/https://www.sketchengine.co.uk/user-guide/user-manual/</ref>. </bibl>
               <bibl>Pescia, Lorenza. 2010. "Il maschile e il femminile nella stampa scritta del
                  Cantone Ticino (Svizzera) e dell’Italia" In <emph>Che genere di lingua? Sessismo e
                     potere discriminatorio delle parole</emph>, edited by Maria Serena Sapegno,
                  57–74. Carocci: Roma.</bibl>
               <bibl>Rossini Favretti, Rema et al. 1998-2017. <emph>CORIS/CODIS. A corpus of written
                     Italian based on a defined and a dynamic model.</emph> Bologna: FICLIT /
                  Università di Bologna. <ref target="https://web.archive.org/web/20170912110322/http://corpora.dslo.unibo.it/coris_eng.html">https://web.archive.org/web/20170912110322/http://corpora.dslo.unibo.it/coris_eng.html</ref>. </bibl>
               <bibl>SSLMIT Dev Online. 2004a. <emph>„La Repubblica“ Corpus. Corpus
                     Information.</emph> Bologna: Università di Bologna. <ref target="https://web.archive.org/web/20170912110839/http://dev.sslmit.unibo.it/corpora/corpus.php?path=&amp;name=Repubblica">https://web.archive.org/web/20170912110839/http://dev.sslmit.unibo.it/corpora/corpus.php?path=&amp;name=Repubblica</ref>. </bibl>
               <bibl>SSLMIT Dev Online. 2004b. <emph>„La Repubblica“ Corpus. Corpus
                     Description.</emph> Bologna: Università di Bologna. <ref target="https://web.archive.org/web/20170912110927/http://dev.sslmit.unibo.it/corpora/corpus.php?path=&amp;name=Repubblica&amp;action=information">https://web.archive.org/web/20170912110927/http://dev.sslmit.unibo.it/corpora/corpus.php?path=&amp;name=Repubblica&amp;action=information</ref>. </bibl>
               <bibl>SSLMIT Dev Online. 2004c. <emph>Frequency Lists How-to (Repubblica).</emph>
                  Bologna: Università di Bologna. <ref target="https://web.archive.org/web/20170912111024/http://dev.sslmit.unibo.it/corpora/frequency_how-to.php?path=&amp;name=Repubblica">https://web.archive.org/web/20170912111024/http://dev.sslmit.unibo.it/corpora/frequency_how-to.php?path=&amp;name=Repubblica</ref>. </bibl>
               <bibl>SSLMIT Dev Online. 2009. <emph>Morph-It!</emph> Version 0.48 (2009-02-23).
                  Bologna: Università di Bologna. <ref target="https://web.archive.org/web/20170912111107/http://dev.sslmit.unibo.it/linguistics/morph-it.php">https://web.archive.org/web/20170912111107/http://dev.sslmit.unibo.it/linguistics/morph-it.php</ref>. </bibl>
               <bibl>TEI Consortium, ed. 2015. <emph>TEI P5: Guidelines for Electronic Text Encoding
                     and Interchange.</emph> Last modified October 4, 2015. <ref target="https://web.archive.org/web/20170912111141/http://www.tei-c.org/Guidelines/P5/">https://web.archive.org/web/20170912111141/http://www.tei-c.org/Guidelines/P5/</ref>.
               </bibl>
            </listBibl>
         </div>
      </back>
   </text>
</TEI>
