<?xml version="1.0" encoding="UTF-8"?>
<?oxygen RNGSchema="../../common/schema/DHQauthor-TEI.rng" type="xml"?>
<?oxygen SCHSchema="../../common/schema/dhqTEI-ready.sch"?>
<TEI xmlns="http://www.tei-c.org/ns/1.0" xmlns:cc="http://web.resource.org/cc/"
   xmlns:dhq="http://www.digitalhumanities.org/ns/dhq"
   xmlns:mml="http://www.w3.org/1998/Math/MathML"
   xmlns:rdf="http://www.w3.org/1999/02/22-rdf-syntax-ns#">
   <teiHeader>
      <fileDesc>
         <titleStmt>
            <!--Author should supply the title and personal information-->
            <title type="article" xml:lang="en">Building open access, calibrated, syntactically
               annotated corpora over the history of French: A Case study of the application and
               adaptation of annotation tools for historical syntax</title>
            <!--Add a <title> with appropriate @xml:lang for articles in languages other than English-->
            <dhq:authorInfo>
               <!--Include a separate <dhq:authorInfo> element for each author-->
               <dhq:author_name>Pierre <dhq:family>Larrivée</dhq:family>
               </dhq:author_name>
               <idno type="ORCID">https://orcid.org/0000-0001-8447-7102</idno>
               <dhq:affiliation>University of Caen</dhq:affiliation>
               <email>pierre.larrivee@unicaen.fr</email>
               <dhq:bio>
                  <p>Pierre Larrivée is professor of Linguistics at the University of Caen,
                     France.</p>
               </dhq:bio>
            </dhq:authorInfo>
            <dhq:authorInfo>
               <!--Include a separate <dhq:authorInfo> element for each author-->
               <dhq:author_name>Natasha <dhq:family>Romanova</dhq:family>
               </dhq:author_name>
               <idno type="ORCID">https://orcid.org/0000-0002-5723-9819</idno>
               <dhq:affiliation>University of Caen</dhq:affiliation>
               <email>natalia.romanova@unicaen.fr</email>
               <dhq:bio>
                  <p>Natasha Romanova is post-doctoral researcher at the CRISCO Lab, , France.</p>
               </dhq:bio>
            </dhq:authorInfo>
            <dhq:authorInfo>
               <!--Include a separate <dhq:authorInfo> element for each author-->
               <dhq:author_name>Mathieu <dhq:family>Goux</dhq:family>
               </dhq:author_name>
               <idno type="ORCID">https://orcid.org/0000-0003-4211-8309</idno>
               <dhq:affiliation>University of Caen</dhq:affiliation>
               <email>mathieu.goux@unicaen.fr</email>
               <dhq:bio>
                  <p>Mathieu Goux is Lecturer in Linguistics at the University of Caen, France.</p>
               </dhq:bio>
            </dhq:authorInfo>
            <dhq:authorInfo>
               <!--Include a separate <dhq:authorInfo> element for each author-->
               <dhq:author_name>Rayan <dhq:family>Ziane</dhq:family>
               </dhq:author_name>
               <idno type="ORCID">https://orcid.org/0009-0000-4940-0169</idno>
               <dhq:affiliation>University of Orléans</dhq:affiliation>
               <email>rayan.ziane@univ-orleans.fr</email>
               <dhq:bio>
                  <p>Rayan Ziane is a PhD student in Linguistics at the University of Orléans,
                     France.</p>
               </dhq:bio>
            </dhq:authorInfo>
            <dhq:authorInfo>
               <!--Include a separate <dhq:authorInfo> element for each author-->
               <dhq:author_name>Morgane <dhq:family>Pica</dhq:family>
               </dhq:author_name>
               <idno type="ORCID">https://orcid.org/0000-0002-0981-4516</idno>
               <dhq:affiliation>University of Caen</dhq:affiliation>
               <email>morgane.pica@unicaen.fr</email>
               <dhq:bio>
                  <p>Morgane Pica is Research Software Engineer at the Pôle Document Numérique,
                     MRSH, University of Caen, France. She is also studying for a PhD in digital
                     history at the Université Paris Sciences et Lettres.</p>
               </dhq:bio>
            </dhq:authorInfo>
         </titleStmt>
         <publicationStmt>
            <publisher>Alliance of Digital Humanities Organizations</publisher>
            <publisher>Association for Computers and the Humanities</publisher>
            <!--This information will be completed at publication-->
            <idno type="DHQarticle-id">000873</idno>
            <idno type="volume">020</idno>
            <idno type="issue">3</idno>
            <date>tbd<!--include @when with ISO date and also content in the form 23 February 2024--></date>
            <dhq:articleType>article</dhq:articleType>
            <availability status="CC-BY-ND">
               <!--If using a different license from the default, choose one of the following:
                  CC-BY-ND (DHQ default):        
                  CC-BY:    
                  CC0:  -->
               <cc:License rdf:about="http://creativecommons.org/licenses/by-nd/2.5/"/>
            </availability>
         </publicationStmt>
         <sourceDesc>
            <p>This is the source</p>
         </sourceDesc>
      </fileDesc>
      <encodingDesc>
         <classDecl>
            <taxonomy xml:id="dhq_keywords">
               <bibl>DHQ classification scheme; full list available at <ref
                     target="https://dhq.digitalhumanities.org/taxonomy.xml"
                     >https://dhq.digitalhumanities.org/taxonomy.xml</ref>
               </bibl>
            </taxonomy>
            <taxonomy xml:id="authorial_keywords">
               <bibl>Keywords supplied by author; no controlled vocabulary</bibl>
            </taxonomy>
            <taxonomy xml:id="project_keywords">
               <bibl>DHQ project registry; full list available at <ref
                     target="https://dhq.digitalhumanities.org/projects.xml"
                     >https://dhq.digitalhumanities.org/projects.xml</ref>
               </bibl>
            </taxonomy>
         </classDecl>
      </encodingDesc>
      <profileDesc>
         <langUsage>
            <language ident="en" extent="default"/>
            <!--add <language> with appropriate @ident for any additional languages-->
         </langUsage>
         <textClass>
            <keywords scheme="#dhq_keywords">
               <!--Authors may suggest one or more keywords from the DHQ keyword list, visible at https://dhq.digitalhumanities.org/taxonomy.xml; these may be supplemented or modified by DHQ editors-->
               <!--Enter keywords below preceeded by a "#". Create a new term element for each-->
               <term corresp="#corpora"/>
               <term corresp="#linguistics"/>
               <term corresp="#language_studies"/>
               <term corresp="#tools"/>
            </keywords>
            <keywords scheme="#authorial_keywords">
               <!--Authors may include one or more keywords (in <term> elements) of their choice-->
               <term>digital corpora</term>
               <term>Lemmatisation</term>
               <term>Automatic syntactic parsint</term>
               <term>diachroning linguistics</term>
            </keywords>
            <keywords scheme="#project_keywords">
               <list type="simple">
                  <item/>
               </list>
            </keywords>
         </textClass>
      </profileDesc>
      <revisionDesc>
         <!-- Replace both "NNNNNN"s in the @target of ther <ref> below with the appropriate DHQarticle-id value. -->
         <change>The version history for this file can be found on <ref type="gitHist"
               target="https://github.com/Digital-Humanities-Quarterly/dhq-journal/commits/main/articles/000873/00873.xml"
               >GitHub</ref>.</change>
      </revisionDesc>
   </teiHeader>
   <text xml:lang="en" type="default">
      <front>
         <dhq:abstract>
            <!--Include a brief abstract of the article-->
            <p>This case study presents challenges and solutions for building calibrated
               syntactically annotated diachronic corpora of French. Through a series of integrated
               projects that developed methods for syntactic annotation of medium and large
               collections of non-literary texts over a long period of time (13th – 19th centuries),
               we consider how the availability of tools and the emergence of research questions
               shape and direct corpus-building activity. Supported by the experience of an earlier
               project, a semi-automated workflow for syntactic annotation was developed to allow
               processing of two corpora calibrated by provenance and genre, with a focus on
               parts-of-speech tagging in three annotation frameworks. The workflow combines the use
               of a neural dependency parser, a lexicon of lemmata and forms and a Python library.
               This experience, in turn, gave rise to a new initiative that aimed at formulating and
               testing best practices for creating syntactically annotated <term>treebanks</term>
               for different periods of the history of the language.</p>
         </dhq:abstract>
         <dhq:teaser>
            <!--Include a brief teaser, no more than a phrase or a single sentence-->
            <p>A case study exploring the building of a calibrated syntactically annotated
               diachronic corpora of French, focusing on collections of non-literary texts between
               the thirteenth and nineteenth centuries.</p>
         </dhq:teaser>
      </front>
      <body>
         <div>
            <head>Introduction<note> An earlier version of this paper was presented at the <quote
                     rend="inline">Journée d’études Retours d’expériences en édition numérique des
                     textes</quote>, organised by Pierre-Jean Suriac, Angela Goebel and Morgane Pica
                  on 22 June 2023 at Université Jean Moulin Lyon 3.</note>
            </head>
            <p>The study of the evolution of language over several centuries is facilitated by
               access to digital collections of syntactically annotated texts. Such
                  <term>corpora</term> allow searching for syntactic structures automatically in
               order to identify particular configurations. Annotation of lemmata, parts of speech
               (PoS) and syntactic functions make it possible for researchers to compare and
               contrast the shape of language through time, explain the observable patterns and test
               general hypotheses. Such enrichment by lexical and grammatical metadata can be
               achieved by modern computer-assisted methods, including, most recently, deep-learning
               tools such as neural network parsers. For the history of French language, a number of
               annotated resources exist. However, they: </p>
            <list type="ordered">
               <item>are usually limited to a particular period, which impedes comparative work
                  across the history of the language; <note> One notable exception is the Frantext
                     corpus that, as of June 2026, contained 275 million words in French from the
                     10th to the 21st centuries from a range of genres. The corpus is lemmatized and
                     annotated in parts of speech (<ref target="https://www.frantext.fr/"
                        >https://www.frantext.fr/</ref>, accessed 1 June 2026). Among existing
                     resources for the medieval period, we can cite the Base du Français Médiéval
                     (URL: <ref target="http://bfm.ens-lyon.fr/">http://bfm.ens-lyon.fr/</ref>,
                     accessed 8 June 2026 <ptr target="#guillotbarbance_etal2017"/>, MCVF-PPCHF <ptr
                        target="#martineau_etal2021"/> and PhraseoRoche <ptr
                        target="#denoyelle_etal2024"/>. For the 16th-17th centuries, SERMO
                        <term>corpus</term> (URL: <ref target="http://sermo.unine.ch/SERMO/"
                        >http://sermo.unine.ch/SERMO/</ref>, accessed 8 June 2026).</note></item>

               <item> consist predominantly of literary material which exhibits conservative traits
                  and is believed to be removed from the everyday language <ptr
                     target="#balon_etal2016"/>
                  <ptr target="#pinzin_etal"/>;</item>
               <item> lack accurate annotation of syntactic functions, which means that even after
                  the extraction of desired configurations of parts of speech considerable
                  investment is often needed to review the data manually.<note> For Old and Middle
                     French two important exceptions are two <term>corpora</term> of literary texts,
                     MCFV-PPCHF <ptr target="#martineau_etal2021"/>, annotated according the UPenn
                     syntactic framework, and Profiterole (formerly known as SRCMF) corpus annotated
                     in Universal Dependencies (UD) <ptr target="#prevost_etal2013"/>
                     <ptr target="#prevost_etal2024"/>.</note>
               </item>
            </list>
            <p>In the present contribution, we discuss how a series of integrated projects conducted
               in 2018-2024 at the CRISCO Laboratory at the University of Caen (France) set out to
               address these challenges making use of language processing technologies available to
               researchers at each stage.<note> The methodological and practical challenges
                  associated with the processing of the textual data from digitization to enrichment
                  with annotation and the solutions proposed are presented in detail in the
                  Methodology section.</note>
            </p>
            <div>
               <head>Corpora</head>
               <p>Supporting research into the history of French syntax, the three projects
                  presented here sought to integrate existing tools and develop methodologies for
                  successful management and syntactic annotation of medium to large diachronic
                     <term>corpora</term>. To limit potential documentation biases, all three share
                  a common approach to the selection of data, opting for texts of a given
                  non-literary genre and produced in the same geographical area (Normandy), to
                  control for extraneous variation (<ref target="#table01">Table 1</ref>). The ConDÉ
                  project (2018-2021)<note> ConDÉ: CONstitution d’un Droit europÉen: six siècles de
                     coutumiers normands <ptr target="#Goux_etal2019"/>. The project data and
                     software have been deposited on GitHub <ref
                        target="https://github.com/RIN-ConDE/editions"
                        >https://github.com/RIN-ConDE/editions</ref> (accessed 8 June 2026). A
                     revised version of the ConDÉ corpus is currently under construction as part of
                     a PhD thesis <ptr target="#pica_inprogress"/>.</note> produced a large
                  digitized corpus of Norman customal law texts (<term>coutumiers normands</term>)
                  from the 13th to the 18th century, based on a method trialed on a set of
                  historical letters (EPELE).<note> The EPELE (Écriture des peu lettrés: Français
                     vernaculaire dans la Normandie médiévale) corpus can be consulted via the
                     project website <ref target="https://pdn-lingua.unicaen.fr/epele/epele/accueil"
                        >https://pdn-lingua.unicaen.fr/epele/epele/accueil</ref> (accessed 8 June
                     2026). </note> Building on the experience of ConDÉ, High-Tech project
                  (2021-2023) gave rise to a collection of chronicles and historical treatises from
                  the 12th to the 19th centuries, with one text per century.<note> High-Tech:
                     High-level text annotation across historical texts: improving
                     semi-automatization of big data management. URL: <ref
                        target="https://www.unicaen.fr/projet_de_recherche/high-tech/"
                        >https://www.unicaen.fr/projet_de_recherche/high-tech/</ref> (accessed 8
                     June 2026). </note> In addition, as part of the Franco-German MICLE project
                  (2021-2024), a further medium-sized corpus of legal texts was elaborated which
                  includes trial accounts and <term>styles de procéder</term> from the 12th to the
                  17th centuries, with one text per half-century.<note> The Franco-German project
                     MICLE (Micro-Cues of Language Evolution) addressed the question of the change
                     of syntactic structure in two Romance language varieties, French of Normandy
                     and Venetian from the origins to the 17th century (URL: <ref
                        target="https://www.unicaen.fr/projet_de_recherche/micle/m"
                        >https://www.unicaen.fr/projet_de_recherche/micle</ref>, accessed 8 June
                     2025). The German team at the University of Frankfurt (Cecilia Poletto and
                     Francesco Pinzin) created an annotated corpus of medieval and early modern
                     Venetian legal texts annotated in UPenn framework. This Venetian corpus
                     (MICLE-VEC) is partially available via the CRISCO Lab TXM portal (URL: <ref
                        target="https://txm-crisco.huma-num.fr/txm/"
                        >https://txm-crisco.huma-num.fr/txm/</ref>, accessed 8 June 2026). See <ptr
                        target="#Goux_etal2024"/>
                     <ptr target="#Goux2024b"/>.</note> Below, we refer to the corpus of customal
                  law as <term>ConDÉ</term>, to the corpus of chronicles produced as part of the
                  High-Tech project as <term>Chroniques</term> and to the corpus of trial accounts
                  and <term>styles de procéder</term>, created by the French team of the MICLE
                  project, as <term>MICLE-Fr</term>. The semi-automated workflow for lemmatization
                  and annotation developed by the High-Tech project and tested during the annotation
                  of <term>Chroniques </term>and <term>MICLE-Fr corpora </term>is referred to as the
                  HT-CRISCO workflow. The annotated collections can be consulted via the CRISCO
                  Laboratory’s TXM server and two dedicated websites.<note> ConDÉ project website:
                        <ref target="https://mrsh.unicaen.fr/coutumiers/conde/accueil.html"
                        >https://mrsh.unicaen.fr/coutumiers/conde/accueil.html</ref>; consultation
                     website for the <term>Chroniques </term>and <term>MICLE-Fr corpora</term> URL:
                        <ref target="https://criscoht.unicaen.fr/"
                        >https://criscoht.unicaen.fr/</ref> (accessed 8 June 2026). </note> Most
                  texts are out of copyright and can be consulted either in full or via a
                  concordance, whereas for a small minority of sources only concordance view is
                  available.</p>
               <table xml:id="table01">
                  <head>Diachronic corpora of Medieval French projects at the CRISCO Lab
                     2018-2024</head>
                  <row role="data">
                     <cell>Project/Funding Period/Funder</cell>
                     <cell>Corpus</cell>
                     <cell>Text Types</cell>
                     <cell>Period</cell>
                     <cell>Number of Tokens</cell>
                     <cell>Annotation</cell>
                  </row>
                  <row role="data">
                     <cell>
                        <p>ConDÉ</p>
                        <p>2018-2021</p>
                        <p>(Réseau d’Intérêts Normands)</p>
                     </cell>
                     <cell>
                        <term>ConDÉ</term>
                     </cell>
                     <cell>Norman customals</cell>
                     <cell>13th-19th cent.</cell>
                     <cell>4,452,540 tokens</cell>
                     <cell>
                        <p>*lemmata</p>
                        <p>*PoS (Presto)</p>
                     </cell>
                  </row>
                  <row role="data">
                     <cell>
                        <p>MICLE</p>
                        <p>2021-2024</p>
                        <p>(ANR/DFG Franco-German grant)</p>
                     </cell>
                     <cell>
                        <term>MICLE-Fr</term>
                     </cell>
                     <cell>
                        <p>Legal texts (trials)</p>
                        <p>Legal treatises (<term>styles de procéder</term>)</p>
                     </cell>
                     <cell>13th-17th cent.</cell>
                     <cell>422,117 tokens</cell>
                     <cell>
                        <p>*lemmata</p>
                        <p>*PoS (UD, UPenn, Presto)</p>
                        <p>*syntactic functions UD (automatic)</p>
                     </cell>
                  </row>
                  <row role="data">
                     <cell>
                        <p>High-Tech</p>
                        <p>2021-2023</p>
                        <p>(Réseau d’Intérêts Normands)</p>
                     </cell>
                     <cell>
                        <term>Chroniques</term>
                     </cell>
                     <cell>
                        <p>Chronicles</p>
                     </cell>
                     <cell>12th-19th cent.</cell>
                     <cell>313,518 tokens</cell>
                     <cell>
                        <p>*lemmata</p>
                        <p>*PoS (UD, UPenn, Presto)</p>
                        <p>*syntactic functions UD (automatic)</p>
                     </cell>
                  </row>
               </table>
            </div>
         </div>
         <div>
            <head>Methodology</head>
            <p>The different stages of processing of the data, from image to annotated text, are
               described below. The major methodological and practical challenges include
               transcription and digitization of large amounts of textual data from heterogeneous
               original sources, lemmatization and syntactic annotation. Appropriate tools need to
               be selected to process the data at each stage. Different types of software use
               different data formats; therefore, conversion scripts need to be provided to ensure
               interoperability. In addition, choices need to be made about when linguists intervene
               to correct outputs of automatic processing. </p>
            <div>
               <head>Transcription</head>
               <p>All corpora were digitized from reproductions of manuscripts or scans of printed
                  editions using OCR and HTR technologies made available via the Transkribus portal.
                     <note> URL: <ref target="https://www.transkribus.org/"
                        >https://www.transkribus.org/</ref> (accessed 8 June 2026). See Nockels et
                     al 2022 <ptr target="#nockels_etal2022"/>. </note> Transkribus allows the use
                  of pre-trained transcription models and training of new models adapted to
                  particular sources. Creating custom-made models was especially necessary in the
                  case of the ConDÉ project, which processed a mass of material: the two volumes by
                  Basnage (1678), for example, run up to a million characters. As part of
                     <term>MICLE-Fr</term>, one particular challenge was the complete and fully
                  verified transcription of a 16th-century manuscript of witness statements from the
                  island of Guernsey Greffe Crime 1 (more than 41,000 tokens).<note> The
                     transcription formed part of an internship project in the spring semester
                     2021-2022 to which student transcribers Agathe Aubert, Lucie-Marie Leblanc,
                     Marie Picard and Valentin Simenel contributed. This source was the object of an
                     MA dissertation in History at the University of Caen <ptr
                        target="#lesquer_2024"/>.</note>
               </p>
               <p>The diversity of sources from across several centuries, ranging from medieval
                  manuscripts to modern critical editions, not to mention early printed books, meant
                  that the texts of the <term>corpora</term> presented considerable variation where
                  spelling of individual words were concerned. The decision to remain as close to
                  the original as possible and not to normalize the spelling was taken for all
                  projects as the tools available permitted to accommodate variation. Lemmatization
                  allowed grouping all grammatical forms and all spelling variants of the same word
                  under the same headword. For example, in the <term>Chroniques</term> corpus, the
                  French word <term>king</term> can be encountered under the forms <term>rei</term>,
                     <term>reis</term>, <term>rey</term>, <term>reys</term>, <term>roi</term>,
                     <term>rois</term>, all lemmatized under the headword <term>roi</term>.</p>
            </div>
            <div>
               <head>File formats</head>
               <p>Once digitized, the texts of <term>ConDÉ</term>, <term>Chroniques</term> and
                     <term>MICLE-Fr corpora</term> were converted into the XML-TEI format <ptr
                     target="#derose1999"/>. This format permits encoding editorial interventions
                  such as corrections of the original (e.g., in case of repetition of words). The
                  use of this format facilitated manual correction of the data throughout the
                  annotation pipeline described below and allowed integration of the final versions
                  of the annotated files into the TXM-portal and websites to enable consultation and
                  interrogation of the <term>corpora</term>. Texts were processed and corrected
                  individually, with a different file corresponding to each.</p>
               <p>The chosen format allowed reflecting the structure of the original (i.e., the
                  manuscript or printed version that had been digitized) in the structure of the XML
                  document. According to the XML-TEI hierarchy, a text can be divided into books, a
                  book can be divided into sections, a section into chapters, and a chapter into
                  paragraphs; the corresponding tags made it possible to indicate where paragraphs
                  and larger sections of the text begin and end.<note> On the importance of not
                     losing sight of the materiality of the original, including the structure of the
                     text, when creating large linguistic <term>corpora</term> see <ptr
                        target="#Goux2024a"/>.</note>
               </p>
               <p>For <term>ConDÉ</term>, only XML-TEI format was used throughout all stages of
                  processing. For <term>Chroniques</term> and <term>MICLE-Fr</term>, files needed to
                  be converted into the CONLL-U tabular format that is used by the syntactic parsers
                  twice during the execution of the workflow (see 3d-2 below). To reintegrate the
                  parsed data into the XML-TEI corpus, CONLL-U files were converted back into the
                  XML-TEI format. The synchronization between the XML-TEI version and CONLL-U
                  version of the files was done via the assignment of a unique ID to each sentence
                  of the text. Thus, sentence 1 in paragraph 3 of chapter 34 in section 1 of book 2
                  would be given sentence ID <quote rend="inline">2_1_34_3_1</quote>. This number
                  would, in turn, allow finding the place of the sentence in the XML-TEI structure
                  after parsing is completed.</p>
            </div>
            <div>
               <head>Segmentation</head>
               <p>Prior to any further treatment, the texts were tokenized and, in the case of
                     <term>Chroniques</term> and <term>MICLE-Fr</term>, segmented into sentences
                  (using strong punctuation as a prompt for automatic segmentation, followed by
                  manual revision). For the ConDÉ project, annotation was done exclusively on the
                  token level, whereas for High-Tech and MICLE-French sentence segmentation was an
                  essential step towards preparing the data for syntactic parsing using an automatic
                  tool.</p>
            </div>
            <div>
               <head>Annotation</head>
               <p>Considering the nature of the data and of the research needs as well as the
                  availability of annotation tools, the sequence of projects followed two different
                  yet complementary axes:</p>
               <p>1) increased accuracy of the annotation and formalization of the correction
                  process;</p>
               <p>2) introduction of more layers of annotation.</p>
               <p>Where <term>ConDÉ</term> is lemmatized and annotated in PoS using one set of tags,
                     <term>Chroniques </term>and <term>MICLE-Fr</term> are lemmatized, PoS-annotated
                  in three different systems, adding a layer of morphological annotation, and
                  (automatically) annotated in syntactic functions. At every stage of this process,
                  new tools, resources and methodological approaches were integrated as they became
                  available to the linguistic community.</p>
               <div>
                  <head>Lemmatization and PoS tagging (<term>ConDÉ</term>)</head>
                  <p>Following an extensive campaign of digitization, conversion to XML-TEI and
                     tokenization, the texts of the <term>ConDÉ</term> corpus were lemmatized and
                     PoS-annotated. These two operations were accomplished using the Presto lexicon.
                     This lexicon takes the form of a list of possible (attested and
                     computer-generated) word forms of modern and historical French
                        <term>lemmata</term>, with an indication of the grammatical category (using
                     the Presto <term>tagset</term>, a set of labels particularly adapted to French)
                     of each word <ptr target="#Blumenthal_etal2017"/>
                     <ptr target="#lay_etal2010"/>.<note> The Presto dictionary is currently
                        available for download in its original (<ref
                           target="https://unicloud.unicaen.fr/index.php/s/NSkPrcaZ3Rx2t9P"
                           >https://unicloud.unicaen.fr/index.php/s/NSkPrcaZ3Rx2t9P</ref>) version
                        and the version revised by the team at CRISCO (<ref
                           target="https://unicloud.unicaen.fr/index.php/s/YgfYJenQMKD8bEC"
                           >https://unicloud.unicaen.fr/index.php/s/YgfYJenQMKD8bEC</ref>) (accessed
                        8 June 2026). </note> The information contained in the lexicon was compared
                     to the data in the XML-TEI file using a dedicated Python script. </p>
                  <p>While the use of the Presto lexicon and the script allowed matching the form of
                     the word with a possible lemma and PoS, it did not make use of contextual
                     information. It neither disambiguated when there was more than one lemma or PoS
                     speech possible, nor offer a solution when the form was not listed. In order to
                     address such cases, a programme of semi-manual correction was put in place to
                     reduce manual intervention. <note> The scripts are available at <ref
                           target="https://github.com/RIN-ConDE/tools/tree/main/corpus-construction/disambiguate-lemmatization-in-corrected-file"
                           >https://github.com/RIN-ConDE/tools/tree/main/corpus-construction/disambiguate-lemmatization-in-corrected-file</ref>
                        (accessed 8 June 2026). </note> For example, the French word form
                        <term>en</term> can have the role of a preposition (usually when followed by
                     a common or proper noun or by a determiner) or that of a pronoun (usually
                     preceding a verb), and disambiguating the two is important for identifying
                     specific kinds of constructions (e.g., prepositional noun phrases or verbs
                     preceded by a pronoun, see <ref target="#figure01">Figure 1</ref>). Similarly,
                     the word <term>fait</term> may be a noun (<term>fact</term>) to be classified
                     under the lemma <term>fait</term> or a form of the verb (<term>to do</term>,
                        <term>to make</term>) to be lemmatized under <term>faire</term>.</p>
                  <figure xml:id="figure01">
                     <head>Example of a concordance search for <term>en</term> in ConDÉ (in the
                        Coutumier by Terrien 1578). The first one is a preposition; the second one
                        is a pronoun.</head>
                     <graphic url="resources/figures/figure01.png"/>
                     <figDesc>Screenshot of two lines of French text with the word "en" highlighted
                        and appearing twice. </figDesc>
                  </figure>
                  <p>Even though, given the size of the corpus (just under four and a half million
                     tokens), a comprehensive manual correction was never conducted, it proved to be
                     a reliable resource for tracing grammatical change <ptr target="#Goux2022"/>.
                  </p>
               </div>
               <div>
                  <head>3d-2. Neural dependency parsing for annotation in PoS, lemmata, syntactic
                     functions (<term>Chroniques</term> and <term>MICLE-Fr</term>)</head>
                  <p>The challenges identified during work on <term>ConDÉ</term> were addressed as
                     part of the elaboration of the annotation pipeline for <term>Chroniques</term>
                     and <term>MICLE-Fr corpora</term>, that resulted from two projects that run in
                     parallel from 2021. Whereas the former focused on methodological issues
                     (creation of an annotation workflow and a user-friendly consultation site), the
                     latter was constituted around the research question that aimed at clarifying
                     the evolution of the Verb Second (V2) word order in a diachronic perspective.
                     The availability of tools and resources, on the one hand, and the research need
                     for simplified access to annotated verb forms and subject and object nominal
                     phrases, on the other, determined the choice of the annotation to be added and
                     the accuracy level of correction. Where in <term>ConDÉ</term> annotation
                     centered on linear succession of the tokens and on a set of rules and
                     light-touch manual interventions to disambiguate ambivalent forms, the
                        <term>HT-CRISCO</term> workflow, also used to annotate
                     <term>MICLE-Fr</term>, put at its core a syntactic parser that operates both at
                     the level of the word and sentence. Graph-based dependency parsers inspired by
                     Dozat and Manning’s architecture <ptr target="#dozat_etal2017"/> are computer
                     programmes that can use models trained on pre-annotated <term>corpora</term> to
                     automatically annotate other texts in PoS (<term>tagging</term>) and syntactic
                     functions (<term>parsing</term>) in the Universal Dependencies (UD) framework
                        <ptr target="#demarneffe_etal2021"/>. The accuracy of annotation depends on
                     the proximity of the language on which the model was trained (<term>training
                        corpus</term>) to the <term>target corpus</term> (corpus to be
                     annotated).</p>
                  <p>The workflow developed and tested by the two project teams relied on a
                     succession of automatic stages followed by manual checks (<ref
                        target="#figure02">Figure 2</ref>).<note> For the full description and
                        Python scripts see <ref
                           target="https://github.com/Corpus-Diachroniques-CRISCO/HT-CRISCO"
                           >https://github.com/Corpus-Diachroniques-CRISCO/HT-CRISCO</ref> (accessed
                        8 June 2026). For further information on the workflow, see Ziane and
                        Romanova, 2024 <ptr target="#ziane_etal2024"/>.</note> The use of an
                     automatic parser allowed to conduct PoS tagging using UD parts of speech to
                     provide a basis and necessary contextual disambiguation for lemmatization and
                     refinement of PoS and morphological information (using UPenn <ptr
                        target="#Santorini2007"/> and Presto <ptr target="#Blumenthal_etal2017"/>
                     <term>tagsets</term>) at later stages of the processing. Manual revision of the
                     data occasionally led to corrections (e.g., of OCR or HTR errors) or changes in
                     sentences segmentation and tokenization, which meant that the automatic parsing
                     (identification of the syntactic head and function of each token) needed to be
                     performed again to take these changes into account (manually verified PoS tags
                     and lemmata were preserved).</p>
                  <p>Among existing dependency parsers, HoPS parser was selected because in 2021 the
                     developers of this tool had made available a model trained on a corpus of Old
                     and Middle French literary texts, and a large proportion of the data to be
                     annotated dated from the medieval period. Contextual analysis provided by HoPS
                     allowed successfully distinguishing between forms that otherwise would have
                     remained ambiguous. To take examples cited before, HoPS was very successful in
                     distinguishing <term>en</term> prepositions (for example <foreign>en
                        Normandie</foreign>
                     <quote rend="inline">in Normandy</quote>) and pronouns (<foreign>et
                        en receipt le roy l’hommaige</foreign>
                     <quote rend="inline">and of-it received the king hommage</quote>, an example
                     from a 1373 text from <foreign>Chroniques</foreign>). Similarly, in <foreign>Je dy qu’il a bien prouvé tel fait</foreign> <quote
                        rend="inline">I say that he proved this fact well</quote> it was possible
                     to correctly tag <foreign>fait</foreign> as a noun and lemmatize under
                     <foreign>fait</foreign> whereas in <foreign>L’adjournement en cas de
                        dolléance est fait en ceste manière</foreign> <quote rend="inline">in case of
                        grievance the adjournment is done in this manner</quote> the same word was
                     tagged as a past participle of the verb <foreign>faire</foreign> (both examples from
                     the 1425 text, <term>MICLE-Fr</term>). The overall performance of automatic
                     parsing on the prediction of parts of speech (<term>tagging</term>) across a
                     very diverse corpus has been very satisfactory.<note> Recent quantitative tests
                        conducted by the team on samples of two texts of the High-Tech corpus using
                        HoPS, UDify and BertForDeprel parsers have shown performances between 87,12%
                        and 90,24% depending on the tool and training conditions.</note>
                  </p>
                  <figure xml:id="figure02">
                     <head>HT-CRISCO Semi-automated workflow for syntactic annotation of diachronic
                        corpora of French (<ref
                           target="https://github.com/Corpus-Diachroniques-CRISCO/HT-CRISCO"
                           >https://github.com/Corpus-Diachroniques-CRISCO/HT-CRISCO</ref>)</head>
                     <graphic url="resources/figures/figure02.png"/>
                     <figDesc>Visual representing the workflow of annotating the corpora. </figDesc>
                  </figure>
               </div>
            </div>
         </div>
         <div>
            <head>Results</head>
            <p>The absence of annotation of syntactic functions in <term>ConDÉ</term> limited
               research potential of the corpus to an extent (in part compensated by the corpus
               size). Thus, the corpus was highly relevant for the search of particular items, such
               as relatively infrequent postverbal clitic pronouns <ptr target="#olivier_etal2023"
               />. Syntactic queries called for more elaborate strategies. An example is the
               analysis of the so-called <term>bare</term> nouns in the corpus <ptr
                  target="#larrivee_etal2024a"/>. Whereas French nouns could be used without an
               article (also known as <term>determiner</term>) in its early history, as is the case
               in other Romance languages and in Latin, that possibility was severely reduced over
               time. Because the search could not refer to functions that would have allowed us to
               pick up the boundaries of the noun phrase, finding nouns without a determiner had to
               be realized using approximate heuristics. The strategy was to look for nouns that
               were not immediately preceded by a determiner. Nouns preceded by an adjective before
               the article were therefore counted as <term>bare</term> (i.e., nouns used without an
               article). We had to be content with the fact that the same search was applied equally
               to all texts, meaning that the same systematic errors would creep in, allowing
               meaningful comparison. The alternative would have been a manual examination of data,
               and, while this was impractical given the thousands of occurrences concerned, that
               could not ensure that the occurrences not returned did not contain some relevant
               examples. </p>
            <p>Modern neural network parsers that provide statistics-based predictions not only for
               annotation in PoS (<term>tagging)</term> but also for syntactic functions
                  (<term>parsing</term>). In the UD approach, parsing consists in the identification
               of a series of asymmetric syntactic relations between two tokens with one being the
                  <term>head</term> and the other the <term>dependent</term>; and in the nature of
               the relation, defined by labels such as <term>nsubj</term> (subject),
                  <term>obj</term> (direct object) or <term>advcl </term>(adverbial clause). Such
               annotation is most susceptible to error, especially when the distance between the
                  <term>training corpus</term> used to create the annotation model and the annotated
               corpus is important (e.g., when the <term>corpora</term> belong to different
               dialects, chronological periods and/or textual genres; for example, models for Modern
               French would be less successful on Medieval French material). One text-internal
               factor that we have observed to multiply mistaken syntactic annotation is sentence
               length <ptr target="#ziane_etal2024"/>
               <ptr target="#daoudi_etal2025"/>. Despite its considerable value for linguistic
               research, syntactic annotation would thus require most resources, time and expertise
               to be corrected. The creation of manually checked <term>treebanks</term> being
               outside the scope of <term>ConDÉ</term>, <term>Chroniques</term> and
                  <term>MICLE-Fr</term> projects, we nevertheless empirically observed that
               automatic annotation in key functions such as subjects and objects was quite reliable
               and, combined with PoS annotation and lemmatization, allows structures such as
               postverbal subjects <ptr target="#larrivee_etal2024b"/> and preposed objects <ptr
                  target="#larrivee2024"/> to be identified and to understand and explain their
               patterns of evolution.<note> In addition, during the annotation process, automatic
                  annotation in functions was used to automatize disambiguation of some verb forms
                  during the conversion from UD to UPENN and Presto <term>tagsets</term>: e.g., if
                  the form <term>fait</term> that can be a finite form or a past participle of the
                  verb <term>faire</term> has an auxiliary <term>avoir</term> or <term>être</term>
                  we consider it to be a past participle <ptr target="#ziane_etal2023a"/>.</note>
            </p>
            <p>At the center of our preoccupations when setting up and implementing the diachronic
                  <term>corpora</term> projects at the CRISCO lab has been the promotion of
               accessibility of processes and alignment to international practice. An example was
               the development as part of the High-TECH project of a visualization and search portal
               to make resources readable and searchable in an ergonomic format.<note>
                  <ref target="https://criscoht.unicaen.fr/">https://criscoht.unicaen.fr/</ref>
                  (accessed 1 June 2026).</note> We adopted UD format and tools to contribute to the
               development of both the historical French corpus field, and the international
               treebank movement, while providing tags in other annotation systems. </p>
            <p>Finally, the annotation process and the scripts developed as part of
                  <term>HT-CRISCO</term> workflow can be used for lemmatization and PoS-tagging of
               new diachronic <term>corpora</term> in French to achieve annotation that can be used
               for exploring texts of any period of the history of the language.</p>
         </div>
         <div>
            <head>Research perspectives</head>
            <p>Broader implications of the projects discussed here are found in highlighting
               shortcomings in the current approaches to <term>corpora</term> building. The combined
               <foreign>Chroniques</foreign> and <term>MICLE-Fr corpora</term> result in a collection
               of just under a million tokens across eight centuries, reflecting diachronic
               variation in the calibrated conditions of controlled provenance and text type,
               lemmatized and annotated in PoS. The next stage is achieving reliable annotation of
               syntactic functions to enable cross-textual analysis that can be done <term>in one
                  click</term> without a need for approximation or manual post-processing. In
               addition to the costly demands in human effort to correct large <term>corpora</term>,
               we are faced with two main interconnected challenges: </p>
            <list type="ordered">
               <item>Lack of training <term>corpora</term> for creating models for all periods of
                  the history of the French language. <p>The texts of <foreign>Chroniques</foreign> and
                        <term>MICLE-Fr</term> present considerable diachronic variation, ranging as
                     they are from the 12th to the 19th century. As far as the choice of annotation
                     model is concerned, in the UD collection, only the medieval and the
                     contemporary period have reference <term>corpora</term> for French
                        language.<note> For the comparative statistics of the Modern French UD
                        treebanks, see <ref
                           target="https://universaldependencies.org/treebanks/fr-comparison.html"
                           >https://universaldependencies.org/treebanks/fr-comparison.html</ref>
                        (accessed 8 June 2026). Profiterole Old French and Profiterole Middle French
                           <term>corpora</term> (<ref
                           target="https://universaldependencies.org/treebanks/fro_profiterole/index.html"
                           >https://universaldependencies.org/treebanks/fro_profiterole/index.html</ref>
                        and <ref target="https://universaldependencies.org/frm/index.html"
                           >https://universaldependencies.org/frm/index.html</ref>, accessed 8 June
                        2026) were formerly known as SRCMF <ptr target="#prevost_etal2013"/>. HoPS
                        parser has a pre-trained model based on the previous version of the Old and
                        Middle French corpus (<ref
                           target="https://github.com/hopsparser/hopsparser/blob/main/docs/models.md"
                           >https://github.com/hopsparser/hopsparser/blob/main/docs/models.md</ref>,
                        accessed 8 June 2026).</note>
                  </p></item>
               <item>Lack of consistency in the annotation principles for Medieval and Modern French
                  which impedes training of models and analysis. In the UD collection, Old/Middle
                  French and Modern French are treated as different languages with slightly
                  divergent annotation guidelines, e.g., where the annotation of modal verbs is
                  concerned. For example, for Old/Middle French, modal verbs (<foreign>devoir</foreign>,
                  <foreign>pouvoir</foreign>, <term>souloir</term>) <note> In the Profiterole corpus,
                        <foreign>vouloir</foreign> is no longer considered as a modal
                     verb/auxiliary.</note> are treated as auxiliaries of the main lexical verb in
                  the infinitive (e.g., <foreign>Il peut travailler</foreign> <quote
                     rend="inline">He can work</quote>). For modern French, on the other hand, the
                  modal verb is the head of the clause, and the infinitive of the lexical verb is
                  the head of an open clausal complement (<term>xcomp</term>). For reasons of
                  consistency of annotation and given that the majority of our texts dates from before the
                  17th century, we used the pre-trained model for Old French supplied by the
                  creators of the HoPS parser <ptr target="#grobol_etal2022"/> and we have thus
                  annotated modal verbs as auxiliaries across the corpus but more discussion about
                  homogenizing guidelines for different periods of French (and other Romance
                  languages) is needed.<note>
                     <ref target="https://zenodo.org/records/7708976"
                        >https://zenodo.org/records/7708976</ref> (accessed 8 June
                  2026).</note></item>
            </list>


            <p>In addition to HoPS, several graph-based syntactic parsers are currently available to
               researchers, including UDify <ptr target="#straka_etal2016"/>
               <ptr target="#guiller2020"/>. UDify can be used via a user-friendly portal UDPipe and
               BertForDeprel is integrated into an annotation and graph rewriting tools ArboratorGrew<note>
                  <ref target="https://grew.fr/">https://grew.fr/</ref> (accessed 8 June
                  2026).</note>. The latter also allows retraining models and adapting them to the
               text or corpus being annotated, by progressively adding annotated material to improve
               the quality of the annotation at each iteration, using a <term>bootstrapping</term>
               approach <ptr target="#peng_etal2022"/>
               <ptr target="#romanova_etal2025a"/>. <note> Outside the scope of the present
                  contribution, building on the experience of previous corpus-building initiatives,
                  Automated project (2023-2025), led by Pierre Larrivée and funded by Normandy
                  Region, set out to test the bootstrapping approach to some of the texts of the
                     <term>MICLE-Fr</term> corpus (on recent experiments on adapting models for one
                  text of the <term>MICLE-Fr corpus</term>, see <ptr target="#ziane_etal2024"/>).
                  Two small treebanks of Old and Middle French (URL: <ref
                     target="https://github.com/UniversalDependencies/UD_Old_French-ALTM"
                     >https://github.com/UniversalDependencies/UD_Old_French-ALTM</ref> and <ref
                     target="https://github.com/UniversalDependencies/UD_Middle_French-ALTM"
                     >https://github.com/UniversalDependencies/UD_Middle_French-ALTM</ref>, accessed
                  8 June 2025) were produced as well as larger treebank of 16th-century French (URL:
                     <ref target="https://github.com/UniversalDependencies/UD_French-ALTS"
                     >https://github.com/UniversalDependencies/UD_French-ALTS</ref>, accessed 8 June
                  2026). On the experience of syntactic annotation of an Old Gascon treebank without
                  a preexisting model, see <ptr target="#romanova_etal2025b"/>, </note>
            </p>
         </div>
         <div>
            <head>Funding</head>
            <p>Projects presented in this paper have received funding from Normandy Region (RIN:
               Réseau d’Intérêts Normands): ConDÉ, High-Tech, and from the ANR: Agence Nationale de
               Recherche (<term>MICLE-Fr</term> as part of the ANR-DFG-funded Franco-German scheme).
               Support was also brought by the Institut universitaire de France.</p>
         </div>
      </body>
      <back>
         <listBibl>
            <bibl xml:id="Blumenthal_etal2017" label="Blumenthal et al 2017">Blumenthal, P.,
               Diwersy, S., Falaise, A., Lay, H., Souvay, G. and Vigier, D. (2017) <title
                  rend="quotes">Presto, un corpus diachronique pour le français des XVIe-XXe
                  siècle.</title>
               <title rend="italic">Traitement Automatique des Langues Naturelles (TALN)</title>.
               June 2017, Orléans, France. 18-26. URL: <ref
                  target="https://taln2017.cnrs.fr/wp-content/uploads/2017/06/actes_ACor4French_2017.pdf"
                  >https://taln2017.cnrs.fr/wp-content/uploads/2017/06/actes_ACor4French_2017.pdf</ref>
               (accessed 8 June 2026).</bibl>
            <bibl xml:id="balon_etal2016" label="Balon and Larrivée 2016">Balon, L. and Larrivée, P.
               (2016) <title rend="quotes">L’ancien français n’est déjà plus une langue à sujet nul
                  – nouveau témoignage des textes légaux</title>. <title rend="italic">Journal of
                  French Language Studies</title> 26(2): 221-237.</bibl>


            <bibl xml:id="daoudi_etal2025" label="Daoudi et al 2026">Daoudi, K., Dehouck M., Ziane
               R. and Romanova N. (2025) <title rend="quotes">Explicit Edge Length Coding to Improve
                  Long Sentence Parsing Performance.</title>
               <title rend="italic">Proceedings of the First Workshop on Advancing NLP for
                  Low-Resource Languages</title> Varna, Bulgaria, 102-110. URL: <ref
                  target="https://aclanthology.org/2025.lowresnlp-1.11"
                  >https://aclanthology.org/2025.lowresnlp-1.11</ref> (accessed 8 June 2026).</bibl>

            <bibl xml:id="demarneffe_etal2021" label="de Marneffe et al 2021" sortKey="Demarneffe">de Marneffe, M.-C.,
               Manning, C. D., Nivre, J. and Zeman, D. (2021) <title rend="quotes">Universal
                  Dependencies,</title>
               <title rend="italic">Computational Linguistics</title>, 47(2): 255–308. URL: <ref
                  target="https://doi.org/10.1162/coli_a_00402"
                  >https://doi.org/10.1162/coli_a_00402</ref> (accessed 8 June 2026).</bibl>

            <bibl xml:id="denoyelle_etal2024" label="Denoyelle et al 2024">Denoyelle, C., Kraif, O.,
               Mounier, P., Renwick, A., Sorba, J. and Souvay, G. (2024) <title rend="quotes">Le
                  corpus PhraséoRoChe: les défis de l’établissement des textes et de l’hétérogénéité
                  des états de la langue.</title>
               <title rend="italic">Corpus</title>, 25. 18 pp. URL:<ref
                  target="https://journals.openedition.org/corpus/8501"> </ref>
               <ref target="https://journals.openedition.org/corpus/8501"
                  >https://journals.openedition.org/corpus/8501</ref> (accessed 8 June 2026).</bibl>

            <bibl xml:id="derose1999" label="DeRose 1999">DeRose, S. (1999) <title rend="quotes">XML
                  and the TEI.</title> Computers and the Humanities, 33(1): 11-30. <ref
                  target="https://doi.org/10.1023/A:1001771114509"/></bibl>

            <bibl xml:id="dozat_etal2017" label="Dozat and Manning 2017">Dozat T.C. and Manning, C.
               D. (2017) <title rend="quotes">Deep Biaffine Attention for Neural Dependency
                  Parsing.</title> <title rend="italic">International Conference on Learning Representations (ICLR)</title> 8 pp.
               URL: <ref target="https://arxiv.org/abs/1611.01734"
                  >https://arxiv.org/abs/1611.01734</ref> (accessed 8 June 2026).</bibl>

            <bibl xml:id="Goux_etal2019" label="Goux et al 2019">Goux, M. and Pica, M. (2019) <title
                  rend="quotes">Le projet ConDÉ : présentation. Les défis d’un corpus de textes en
                  diachronie longue.</title>. Conference slides. URL : <ref
                  target="https://hal.science/hal-02447030"> </ref>
               <ref target="https://hal.science/hal-02447030">https://hal.science/hal-02447030</ref>
               (accessed 8 June 2026).</bibl>
            <bibl xml:id="Goux2022" label="Goux 2022">Goux, M. (2022) <title rend="quotes">Le temps
                  long : l'évolution du français dans un corpus textuel calibré. Le témoignage de la
                  coutume de Normandie.</title>
               <title rend="italic">Studia Linguistica Romanica</title>, 8. URL: <ref
                  target="https://studialinguisticaromanica.org/index.php/slr/article/view/99"
                  >https://studialinguisticaromanica.org/index.php/slr/article/view/99</ref>
               (accessed 8 June 2026)</bibl>
            <bibl xml:id="Goux_etal2024" label="Goux and Pinzin 2024">Goux, M. and Pinzin, F. (2024)
                  <title rend="quotes">Challenges of a Multilingual Corpus (Old French/Old
                  Venetian): The Example of the MICLE project.</title> Fontana A. and Pezzini E.
               Franco Cesati (eds.) <title rend="italic">Venezia e la Francia tra Medioevo ed età
                  Moderna. Similitudini, specificità, interrelazioni</title>. Florence : Cesati
               Editore, pp. 153-175.</bibl>

            <bibl xml:id="Goux2024a" label="Goux 2024(a)">Goux, M. (2024) <title rend="quotes">De
                  très grands corpus pour l’étude diachronique du français : annotations,
                  informations métalinguistiques et paratextes.</title>
               <title rend="italic">Humanités numériques</title>, 9. <ref
                  target="https://journals.openedition.org/revuehn/3930"
                  >https://journals.openedition.org/revuehn/3930</ref> (accessed 8 June
               2026).</bibl>


            <bibl xml:id="Goux2024b" label="Goux 2024(b)">Goux, M. (2024) <title rend="quotes"
                  >Enjeux des corpus bilingues en diachronie longue : l’exemple du projet
                  MICLE.</title> <title rend="italic">Corpus</title>, 25. 13 pp. URL: <ref
                  target="https://journals.openedition.org/corpus/8468"
                  >https://journals.openedition.org/corpus/8468</ref> (accessed 8 June 2026). </bibl>

            <bibl xml:id="grobol_etal2022" label="Grobol et al 2022">Grobol, L., Regnault, M., Ortiz
               Suarez, P., Sagot, B., Romary, L., and Crabbé, B. (2022). <title rend="quotes"
                  >BERTrade: Using Contextual Embeddings to Parse Old French.</title> IN Calzolari,
               N., Béchet, F., Blache, P., Choukri, K., Cieri, C., Declerck, C., Goggi, S., Isahara,
               H., Maegaard, B., Mariani, J., Mazo, H., Odijk, J., and Piperidis, S. <title
                  rend="italic">Proceedings of the Thirteenth Language Resources and Evaluation
                  Conference</title> European Language Resources Association, pp. 1104–1113. URL:
                  <ref target="https://aclanthology.org/2022.lrec-1.119"
                  >https://aclanthology.org/2022.lrec-1.119</ref> (accessed 8 June 2026). </bibl>

            <bibl xml:id="guiller2020" label="Guiller 2020">Guiller, K. (2020). <title rend="italic"
                  >Analyse syntaxique automatique du pidgin-créole du Nigeria à l’aide d’un
                  transformer (BERT): Méthodes et Résultats.</title> MA Dissertation, Sorbonne
               Nouvelle.</bibl>


            <bibl xml:id="guillotbarbance_etal2017" label="Guillot-Barbance et al 2017"
               >Guillot-Barbance, C., Heiden, S. and Lavrentiev, A. (2017) <title rend="quotes">Base
                  de français médiéval : une base de référence de sources médiévales ouverte et
                  libre au service de la communauté scientifique.</title> Diachroniques, 7:168-184.
               URL: <ref target="https://shs.hal.science/halshs-01809581"
                  >https://shs.hal.science/halshs-01809581</ref> (accessed 8 June 2026).</bibl>

            <bibl xml:id="larrivee2024" label="Larrivée 2024" sortKey="Larrivee">Larrivée, P. (2024)
                  <title rend="quotes">Deux déterminismes du déclin des objets antéposés dans
                  l'histoire du français.</title> Congrès Mondial de Linguistique Française, 2024,
               Université de Lausanne.</bibl>
            <bibl xml:id="larrivee_etal2024a" label="Larrivée and Goux 2024" sortKey="Larrivee"
               >Larrivée, P. and Goux, M. (2024).<title rend="quotes">The evolution of bare nouns in
                  the history of French. The view from calibrated corpora.</title> 34(4), pp. 323 -
               350. URL: <ref
                  target="https://www.cambridge.org/core/journals/journal-of-french-language-studies/article/evolution-of-bare-nouns-in-the-history-of-french-the-view-from-calibrated-corpora/A809D923F7CD0854DBB9833DFAB4C428"
                  >https://www.cambridge.org/core/journals/journal-of-french-language-studies/article/evolution-of-bare-nouns-in-the-history-of-french-the-view-from-calibrated-corpora/A809D923F7CD0854DBB9833DFAB4C428</ref>
               (accessed 8 June 2026).</bibl>
            <bibl xml:id="larrivee_etal2024b" label="Larrivée et al 2024" sortKey="Larrivee"
               >Larrivée, P., Poletto, C., Pinzin, F., and Goux, M. (2024) <title rend="quotes"
                  >Asymmetry as a general cue for V2 (loss).</title> Isogloss: Open Journal of
               Romance Linguistics, 10(7). 22 p. URL: <ref target="https://hal.science/hal-04839429"
                  >https://hal.science/hal-04839429</ref> (accessed 8 June 2026).</bibl>

            <bibl xml:id="lay_etal2010" label="Lay and Pincemin 2010">Lay, M.-H. and Pincemin, B.
               (2010) <title rend="quotes">Pour une exploration humaniste des textes:
                  AnaLog.</title>. Proceedings of 10th International Conference Journée d’Analyse
               statistique des Données Textuelles 9-11 Juin 2010 – Sapienza University of Rome.
               Bolasco, S., Chiari I. and Giuliano L. (eds.) V.2, 1045-1056.</bibl>

            <bibl xml:id="martineau_etal2021" label="Martineau et al 2021">Martineau, F.,
               Hirschbühler P., Kroch A. et Morin Y.Ch. (2021) MCVF Corpus, parsed, version 2.0.
               URL: <ref target="https://github.com/beatrice57/mcvf-plus-ppchf"
                  >https://github.com/beatrice57/mcvf-plus-ppchf</ref> (accessed 8 June
               2026).</bibl>

            <bibl xml:id="lesquer_2024" label="Le Squer 2024">Le Squer M. (2024) <title
                  rend="quotes">Le registre « Crime » 1563-1569 au Greffe de Guernesey.</title>.
               Master’s dissertation. University of Caen.</bibl>


            <bibl xml:id="nockels_etal2022" label="Nockels et al 2022">Nockels J., Gooding P., Ames,
               S. and Terras M. (2022) <title rend="quotes">Understanding the application of
                  handwritten text recognition technology in heritage contexts: A systematic review
                  of Transkribus in published research.</title> Archival Science, 22(3):
               267-392.</bibl>


            <bibl xml:id="olivier_etal2023" label="Olivier and Folli 2023">Olivier, M., Sevdali, C.
               and Raffaella, F. (2023) <title rend="quotes">Clitic climbing and restructuring in
                  the history of French.</title>
               <title rend="italic">Glossa: a journal of general linguistics</title>, 8(1):
               1–45.</bibl>

            <bibl xml:id="peng_etal2022" label="Peng et al 2022">Peng, Z., Gerdes, K., and Guiller,
               K. (2022).<title rend="quotes">Pull your treebank up by its own bootstraps</title> in
                  Becerra L., Favre B., Gardent C., and Parmentier Y. (eds.) <title
                  rend="italic">Journées Jointes des Groupements de Recherche Linguistique
                  Informatique, Formelle et de Terrain (LIFT) et Traitement Automatique des Langues
                  (TAL)</title>. Pp. 139–153. URL: <ref target="https://hal.science/hal-03846834"
                  >https://hal.science/hal-03846834</ref> (accessed 8 June 2026).</bibl>


            <bibl xml:id="pica_inprogress" label="Pica in preparation">Pica, M. (in preparation)
                  <title rend="italic">L'intertextualité dans les commentaires sur la coutume de
                  Normandie, miroir de l'univers intellectuel des praticiens du droit à l'époque
                  moderne. Apprentissage automatique et triplets RDF au service de la recherche en
                  Histoire</title>. (Thesis in preparation at the Université Paris Sciences et
               Lettres under the supervision of Arabeyre P. and Hodel T.) URL: <ref
                  target="https://theses.fr/s378092">https://theses.fr/s378092</ref> (accessed 8
               June 2026).</bibl>

            <bibl xml:id="pinzin_etal" label="Pinzin and Goux forthcoming">Pinzin, F. and Goux, M.
                  <title rend="quotes">How genre affects word order: a diachronic analysis of
                  French</title>, in P. Larrivée and F. Pinzin (eds.) <title rend="italic">Syntactic
                  change through text-types.</title> Berlin: de Gruyter.</bibl>

            <bibl xml:id="prevost_etal2013" label="Prévost and Stein 2013" sortKey="Prevost"
               >Prévost, S. and Stein, A. (2013) <title rend="quotes">Syntactic annotation of
                  medieval texts: the Syntactic Reference Corpus of Medieval French (SRCMF)</title>,
               in Bennett, P., Durrell, M., Scheible, S. and Whitt, R. (eds.) <title rend="italic"
                  >New Methods in Historical Corpus Linguistics</title>. Narr Verlag.
               Pp.275-282.</bibl>

            <bibl xml:id="prevost_etal2024" label="Prévost et al 2024 Prévost" sortKey="Prevost">
               Prévost, S., Grobol, L., Dehouck, M., Lavrentiev, A. and Heiden. S. (2024) <title
                  rend="quotes">Profiterole: un corpus morpho-syntaxique et syntaxique de français
                  médiéval.</title> <title rend="italic">Corpus</title> 25. 25 pp. URL: <ref
                  target="https://journals.openedition.org/corpus/8538"
                  >https://journals.openedition.org/corpus/8538</ref> (accessed 8 June 2026)</bibl>

            <bibl xml:id="romanova_etal2025a" label="Romanova et al 2025(a)">Romanova, N., Larrivée,
               P. and Ziane, R. (2025) <title rend="italic">Procedure for semi-automatic parsing of
                  Romance corpora (Version 1).</title> Zenodo. URL: <ref
                  target="https://doi.org/10.5281/zenodo.17737727"
                  >https://doi.org/10.5281/zenodo.17737727</ref> (accessed 8 June 2026).</bibl>

            <bibl xml:id="romanova_etal2025b" label="Romanova et al 2025(b)">Romanova, N., Ziane, R.
               and Francioni, B. (2025) <title rend="quotes">Adaptation of models for parsing of Old
                  Gascon.</title> <title rend="italic">Lift2-2025: Journées scientifiques du réseau
                  thématique LIFT2 – linguistique informatique, formelle et de terrain, GDR
                  LIFT.</title> URL: <ref target="https://hal.science/hal-05338944"
                  >https://hal.science/hal-05338944</ref> (accessed 8 June 2026).</bibl>

            <bibl xml:id="Santorini2007" label="Santorini 2007">Santorini, B. (2007) <title
                  rend="quotes">Protocole d'étiquetage - Parties du discours (PDD).</title>
               <ref
                  target="https://www.ling.upenn.edu/~beatrice/corpus-ling/annotation-french/pos/pos-index.html"
                  >https://www.ling.upenn.edu/~beatrice/corpus-ling/annotation-french/pos/pos-index.html</ref></bibl>


            <bibl xml:id="straka_etal2016" label="Straka et al 2016">Straka, M., Hajič, J. and
               Straková, J. (2016) <title rend="quotes">UDPipe: Trainable Pipeline for Processing
                  CoNLL-U Files Performing Tokenization, Morphological Analysis, POS Tagging and
                  Parsing.</title> <title rend="italic">Proceedings of the Tenth International
                  Conference on Language Resources and Evaluation (LREC'16)</title>, European
               Language Resources Association (ELRA), Portorož, Slovenia. pp. 4290–4297.</bibl>

            <bibl xml:id="ziane_etal2023a" label="Ziane and Romanova 2023">Ziane, R. and Romanova,
               N. (2023) <title rend="quotes">Vers l’intégration des outils d’annotation syntaxique:
                  proposition d’une chaîne de traitement itérative pour faciliter l’adoption et
                  l’accès aux technologies d’apprentissage automatique.</title>
               <title rend="italic">Actes des 11èmes Journées Internationales de Linguistique de
                  Corpus, 3-7 juillet 2023</title>, pp. 278-383. URL: <ref
                  target="https://jlc2023.sciencesconf.org/data/pages/abstracts_JLC_2024.pdf"
                  >https://jlc2023.sciencesconf.org/data/pages/abstracts_JLC_2024.pdf</ref>
               (accessed 8 June 2026).</bibl>


            <bibl xml:id="ziane_etal2024" label="Ziane and Romanova 2024">Ziane, R. and Romanova, N.
               (2024) <title rend="quotes">Pistes pour l'optimisation de modèles de parsing
                  syntaxique.</title> <title rend="italic">LIFT 2 - 2024: Journées de
                  lancement</title>, Nov 2024, Orléans, France. URL: <ref
                  target="https://hal.science/hal-04800011v1"
                  >https://hal.science/hal-04800011v1</ref> (accessed 8 June 2026).</bibl>
         </listBibl>
      </back>
   </text>
</TEI>
