<?xml version="1.0" encoding="UTF-8"?>
<resource xsi:schemaLocation="http://datacite.org/schema/kernel-4 http://schema.datacite.org/meta/kernel-4/metadata.xsd"
          xmlns="http://datacite.org/schema/kernel-4"
          xmlns:xsi="http://www.w3.org/2001/XMLSchema-instance">
    <identifier identifierType="DOI">10.34934/DVN/HNA7QO</identifier>
    <creators><creator><creatorName>Zuzana Černáková</creatorName><affiliation>(Department of History, Ghent University; Digital Research Lab, KBR)</affiliation></creator><creator><creatorName>Fien Messens</creatorName><affiliation>(Department of History, Ghent University; KBR)</affiliation></creator><creator><creatorName>Tess Dejaeghere</creatorName><affiliation>(Ghent Centre for Digital Humanities; Language Translation and Technology Team, Ghent University)</affiliation></creator><creator><creatorName>Julie M. Birkholz</creatorName><affiliation>(Ghent Centre for Digital Humanities; Department of History, Ghent University; Digital Research Lab, KBR)</affiliation></creator></creators>
    <titles>
        <title>Replication Data for: From nineteenth-century letters to entities: "a NER pipeline for French correspondence and its methodological lessons" - Article for Digital Humanities Benelux Journal</title>
    </titles>
    <publisher>Social Sciences and Digital Humanities Archive – SODHA</publisher>
    <publicationYear>2026</publicationYear>
    <resourceType resourceTypeGeneral="Dataset"/>
    
    <descriptions>
        <description descriptionType="Abstract">This dataset accompanies the article From nineteenth-century letters to entities: A Named Entity Recognition pipeline for French correspondence and its methodological lessons. It contains the input data, preprocessing scripts, analysis notebooks, evaluation outputs, and experimental results used to develop and evaluate Named Entity Recognition (NER) workflows for nineteenth-century French correspondence from the François-Joseph Navez corpus (KBR – Royal Library of Belgium). The dataset documents the complete experimental workflow, from the construction of a manually annotated gold-standard corpus to the evaluation of off-the-shelf spaCy models, a custom-trained spaCy model, and transformer-based models (CamemBERT, CamemBERTav2, D&apos;AlemBERT and Europeana BERT). The repository is organised into four directories. Input contains the Label Studio annotation exports, the preprocessed gold-standard corpus, and the training, development and test splits used throughout the experiments. Notebooks provides the Jupyter notebooks implementing the preprocessing, training and evaluation workflows. Output contains the evaluation results of the spaCy experiments, including model predictions, overall and per-entity performance metrics, mismatch analyses and summary tables. Results_transformers contains the outputs of the transformer-based experiments, including hyperparameter optimisation, cross-validation results, learning curves, statistical significance tests, prediction files and qualitative error analyses. Together, these files provide the complete computational workflow and all intermediate and final outputs required to reproduce the analyses presented in the accompanying publication.</description>
    </descriptions>
    <contributors><contributor contributorType="ContactPerson"><contributorName>Messens, Fien</contributorName><affiliation>(KBR)</affiliation></contributor></contributors>
</resource>
