@inproceedings{11568_1205508,
 abstract = {In this work, we introduce BureauBERTo, the first transformer-based language model adapted to the Italian Public Administration (PA) and technical-bureaucratic domains. We further pre-trained the general-purpose Italian model UmBERTo on a corpus of PA, banking, and insurance documents, and we expanded UmBERTo's vocabulary with domain-specific terms. We show that BureauBERTo benefitted from the adaptation by comparing it with UmBERTo in both an intrinsic and extrinsic evaluation. The intrinsic evaluation has been conducted through specific fill-mask experiments. The extrinsic one has been faced with a named entity recognition task on one of the sub-domains in BureauBERTo.},
 author = {Auriemma, Serena and Madeddu, Mauro and Miliani, Martina and Bondielli, Alessandro and Passaro, Lucia C. and Lenci, Alessandro},
 booktitle = {Proceedings of the Italia Intelligenza Artificiale - Thematic Workshops Co-Located with the 3rd {{CINI}} National Lab {{AIIS}} Conference on Artificial Intelligence (Ital {{IA}} 2023)},
 keywords = {Domain Adaptation,Evaluation,Italian Bureaucratic Language,NLP,Public Administration,Transformers},
 pages = {240--248},
 publisher = {CEUR-WS.org},
 title = {{{BureauBERTo}}: Adapting {{UmBERTo}} to the {{Italian}} Bureaucratic Language},
 volume = {3486},
 year = {2023}
}

