[{"id":457666,"id_source":589603,"institutes":["ILC"],"type":"journal_article","type_order":1,"title":"What makes a book review compelling? Analyzing informativeness, writing style, and enjoyment","year":2026,"authors":["Alzetta, C.","Dell'Orletta, F.","Miaschi, A.","Venturi, G."],"authors_source":"Alzetta, Chiara; Dell'Orletta, Felice; Miaschi, Alessio; Venturi, Giulia","authors_cnr_name":["ALZETTA, CHIARA","DELL'ORLETTA, FELICE","MIASCHI, ALESSIO","VENTURI, GIULIA"],"authors_cnr_id":["rp12530","rp22811","rp12522","rp00732"],"authors_cnr_institute":[],"abstract":"Amateur book reviews published on Digital Social Reading (DSR) platforms play a crucial role in sharing reading experiences and capturing reader preferences. However, little attention has been given to the linguistic features that influence the way reading experiences are conveyed and shape readers\u2019 perceptions of the aspects discussed in reviews. This study addresses this gap by combining Computational Stylometry and Machine Learning techniques to examine how the linguistic features of book reviews combined with the demographic characteristics of review readers impact the reception of book reviews, focusing in particular on three key aspects that might make a book review compelling, namely informativeness, writing style, and enjoyment. To this aim, we relied on a corpus of Italian Goodreads reviews to investigate how amateur reviewers communicate their reading experience and on a survey to collect human judgments about review reception. Additionally, we investigated the extent to which the review style and the demographic characteristics of review readers can predict the different aspects of review perception. Our findings revealed that linguistic characteristics play a crucial role in shaping reader perceptions and are equally or even more predictive than demographic information in automatic classification models. Nevertheless, while demographic factors such as gender and birth year offer limited utility in forming homogeneous reader groups, reading habits emerged as a relevant factor in identifying shared trends among readers. These insights contribute to a deeper understanding of reader involvement in DSR communities and offer valuable insights for both publishers and review platforms in defining book recommender systems","keywords":["book reviews, human perception, reading experience, stylistic analysis, perception prediction"],"pages":"","url":"https:\/\/doi.org\/10.1093\/llc\/fqag080","volume":"","doi":"10.1093\/llc\/fqag080","editors":[],"editors_source":"","published":"DIGITAL SCHOLARSHIP IN THE HUMANITIES","publisher":"","issn":"2055-7671","isbn":"","conference_name":"","conference_place":"","conference_date":"","last_updated_cnr":"2026-07-09 00:17:49","last_updated_oai":"2026-07-09 00:17:49","last_updated_www":"0000-00-00 00:00:00"},{"id":1098,"id_source":570443,"institutes":["ILC"],"type":"journal_article","type_order":1,"title":"Parallel Trees: a novel resource with aligned dependency and constituency syntactic representations","year":2025,"authors":["Alzetta, C.","Miaschi, A.","Dell'Orletta, F.","Venturi, G.","Montemagni, S."],"authors_source":"Alzetta, C.; Miaschi, A.; Dell'Orletta, F.; Venturi, G.; Montemagni, S.","authors_cnr_name":["ALZETTA, CHIARA","MIASCHI, ALESSIO","DELL'ORLETTA, FELICE","VENTURI, GIULIA","MONTEMAGNI, SIMONETTA"],"authors_cnr_id":["rp12530","rp12522","rp22811","rp00732","rp16780"],"authors_cnr_institute":[],"abstract":"The paper introduces Parallel Trees, a novel multilingual treebank collection that includes 20 treebanks for 10 languages. The distinguishing property of this resource is that the sentences of each language are annotated using two syntactic representation paradigms (SRPs), respectively based on the notions of dependency and constituency. By aligning the annotations of existing resources, Parallel Trees represents an example of exploiting pre-existing treebanks to adapt them to novel applications. To illustrate its potential, we present a case study where the resource is employed as a benchmark to investigate whether and how BERT, one of the first prominent neural language models (NLMs), is sensitive to the dependency-and constituency-based approaches for representing the syntactic structure of a sentence. The case study results indicate that the model's sensitivity fluctuates across languages and experimental settings. The unique nature of the Parallel Trees resource creates the prerequisites for innovative studies comparing dependency and phrase-structure trees, allowing for more focused investigations without the interference of lexical variation","keywords":["Parallel treebanks","Syntactic representation","Diagnostic probing paradigm","Neural language model"],"pages":"3445-3485","url":"https:\/\/iris.cnr.it\/handle\/20.500.14243\/570443","volume":"59 (4)","doi":"10.1007\/s10579-025-09826-3","editors":[],"editors_source":"","published":"LANGUAGE RESOURCES AND EVALUATION","publisher":"","issn":"1574-020X","isbn":"","conference_name":"","conference_place":"","conference_date":"","last_updated_cnr":"2026-03-04 01:27:44","last_updated_oai":"2026-03-04 01:27:44","last_updated_www":"0000-00-00 00:00:00"},{"id":823,"id_source":570522,"institutes":["ILC"],"type":"journal_article","type_order":1,"title":"Leveraging encoder-only large language models for mobile app review feature extraction","year":2025,"authors":["Motger, Q.","Miaschi, A.","Dell'Orletta, F.","Franch, X.","Marco, J."],"authors_source":"Motger, Q.; Miaschi, A.; Dell'Orletta, F.; Franch, X.; Marco, J.","authors_cnr_name":["MIASCHI, ALESSIO","DELL'ORLETTA, FELICE"],"authors_cnr_id":["rp12522","rp22811"],"authors_cnr_institute":[],"abstract":"Mobile app review analysis presents unique challenges due to the low quality, subjective bias, and noisy content of user-generated documents. Extracting features from these reviews is essential for tasks such as feature prioritization and sentiment analysis, but it remains a challenging task. Meanwhile, encoder-only models based on the Transformer architecture have shown promising results for classification and information extraction tasks for multiple software engineering processes. This study explores the hypothesis that encoder-only large language models can enhance feature extraction from mobile app reviews. By leveraging crowdsourced annotations from an industrial context, we redefine feature extraction as a supervised token classification task. Our approach includes extending the pre-training of these models with a large corpus of user reviews to improve contextual understanding and employing instance selection techniques to optimize model fine-tuning. Empirical evaluations demonstrate that these methods improve the precision and recall of extracted features and enhance performance efficiency. Key contributions include a novel approach to feature extraction, annotated datasets, extended pre-trained models, and an instance selection mechanism for cost-effective fine-tuning. This research provides practical methods and empirical evidence in applying large language models to natural language processing tasks within mobile app reviews, offering improved performance in feature extraction","keywords":["Extended pre-training","Feature extraction","Instance selection","Large language models","Mobile app reviews","Named-entity recognition"],"pages":"","url":"https:\/\/iris.cnr.it\/handle\/20.500.14243\/570522","volume":"30 (3)","doi":"10.1007\/s10664-025-10660-y","editors":[],"editors_source":"","published":"EMPIRICAL SOFTWARE ENGINEERING","publisher":"","issn":"1382-3256","isbn":"","conference_name":"","conference_place":"","conference_date":"","last_updated_cnr":"2026-03-04 01:28:12","last_updated_oai":"2026-03-04 01:28:12","last_updated_www":"0000-00-00 00:00:00"},{"id":442,"id_source":570746,"institutes":["ILC"],"type":"conference_article","type_order":7,"title":"The OuLiBench Benchmark: Formal Constraints as a Lens into LLM Linguistic Competence","year":2025,"authors":["Calderaro, S.","Miaschi, A.","Dell'Orletta, F."],"authors_source":"Calderaro, Silvio; Miaschi, Alessio; Dell'Orletta, Felice","authors_cnr_name":["MIASCHI, ALESSIO","DELL'ORLETTA, FELICE"],"authors_cnr_id":["rp12522","rp22811"],"authors_cnr_institute":[],"abstract":"Recent progress in Large Language Models (LLMs) has led to impressive capabilities in Natural Language Generation (NLG). However, standard evaluation benchmarks often focus on surface-level performance and are predominantly English-centric, limiting insights into models\u2019 deeper linguistic competences, especially in other languages. In this paper, we introduce OuLiBench, a novel benchmark inspired by the literary movement OuLiPo, designed to evaluate LLMs\u2019 ability to generate Italian text under explicit linguistic constraints, ranging from morpho-syntactic requirements to creative and structural challenges. Our goal is to assess the extent to which LLMs can understand and manipulate language when guided by specific, sometimes artificial constraints. We evaluate a range of state-of-the-art models in both zero-and few-shot settings, comparing performance across constraint types and difficulty levels. Our results highlight significant variability across models and tasks, shedding light on the limits of controllable text generation and offering a new lens for probing LLMs\u2019 generative and linguistic competence beyond traditional benchmarks","keywords":["Large Language Models, Benchmark, Evaluation, Controllable Text Generation"],"pages":"","url":"https:\/\/iris.cnr.it\/handle\/20.500.14243\/570746","volume":"","doi":"","editors":[],"editors_source":"","published":"Proceedings of the Eleventh Italian Conference on Computational Linguistics (CLiC-it 2025)","publisher":"","issn":"","isbn":"","conference_name":"","conference_place":"","conference_date":"","last_updated_cnr":"2026-03-04 01:27:56","last_updated_oai":"2026-03-04 01:27:56","last_updated_www":"0000-00-00 00:00:00"},{"id":1303,"id_source":570462,"institutes":["ILC"],"type":"conference_article","type_order":7,"title":"Evaluating Lexical Proficiency in Neural Language Models","year":2025,"authors":["Ciaccio, C.","Miaschi, A.","Dell'Orletta, F."],"authors_source":"Ciaccio, C.; Miaschi, A.; Dell'Orletta, F.","authors_cnr_name":["Ciaccio, Cristiano","MIASCHI, ALESSIO","DELL'ORLETTA, FELICE"],"authors_cnr_id":["rp27292","rp12522","rp22811"],"authors_cnr_institute":[],"abstract":"We present a novel evaluation framework designed to assess the lexical proficiency and linguistic creativity of Transformer-based Language Models (LMs). We validate the framework by analyzing the performance of a set of LMs of different sizes, in both mono-and multilingual configuration, across tasks involving the generation, definition, and contextual usage of lexicalized words, neologisms, and nonce words. To support these evaluations, we developed a novel dataset of lexical entries for the Italian language, including curated definitions and usage examples sourced from various online platforms. The results highlight the robustness and effectiveness of our framework in evaluating multiple dimensions of LMs' linguistic understanding and offer an insight, through the assessment of their linguistic creativity, on the lexical generalization abilities of LMs","keywords":["Large Language Models (LLMs)","Interpretability"],"pages":"1267-1286","url":"https:\/\/iris.cnr.it\/handle\/20.500.14243\/570462","volume":"1","doi":"10.18653\/v1\/2025.acl-long.64","editors":[],"editors_source":"","published":"Proceedings of the Annual Meeting of the Association for Computational Linguistics","publisher":"Association for Computational Linguistics (ACL)","issn":"","isbn":"","conference_name":"63rd Annual Meeting of the Association for Computational Linguistics, ACL 2025","conference_place":"","conference_date":"","last_updated_cnr":"2026-03-04 01:27:55","last_updated_oai":"2026-03-04 01:27:55","last_updated_www":"0000-00-00 00:00:00"},{"id":1004,"id_source":570745,"institutes":["ILC"],"type":"conference_article","type_order":7,"title":"Crossword Space: Latent Manifold Learning for Italian Crosswords and Beyond","year":2025,"authors":["Ciaccio, C.","Sarti, G.","Miaschi, A.","Dell'Orletta, F."],"authors_source":"Ciaccio, Cristiano; Sarti, Gabriele; Miaschi, Alessio; Dell'Orletta, Felice","authors_cnr_name":["Ciaccio, Cristiano","MIASCHI, ALESSIO","DELL'ORLETTA, FELICE"],"authors_cnr_id":["rp27292","rp12522","rp22811"],"authors_cnr_institute":[],"abstract":"Answering crossword puzzle clues presents a challenging retrieval task that requires matching linguistically rich and often ambiguous clues with appropriate solutions. While traditional retrieval-based strategies can commonly be used to address this issue, wordplays and other lateral thinking strategies limit the effectiveness of conventional lexical and semantic approaches. In this work, we address the clue answering task as an information retrieval problem exploiting the potential of encoder-based Transformer models to learn a shared latent space between clues and solutions. In particular, we propose for the first time a collection of siamese and asymmetric dual encoder architectures trained to capture the complex properties and relation characterizing crossword clues and their solutions for the Italian language. After comparing various architectures for this task, we show that the strong retrieval capabilities of these systems extend to neologisms and dictionary terms, suggesting their potential use in linguistic analyses beyond the scope of language games","keywords":["Language Games, Crosswords, Semantic Similarity, Embeddings, Natural Language Processing, Information Retrieval"],"pages":"","url":"https:\/\/iris.cnr.it\/handle\/20.500.14243\/570745","volume":"","doi":"","editors":[],"editors_source":"","published":"Proceedings of the Eleventh Italian Conference on Computational Linguistics (CLiC-it 2025)","publisher":"","issn":"","isbn":"","conference_name":"","conference_place":"","conference_date":"","last_updated_cnr":"2026-03-04 01:28:19","last_updated_oai":"2026-03-04 01:28:19","last_updated_www":"0000-00-00 00:00:00"},{"id":1772,"id_source":570461,"institutes":["ILC"],"type":"conference_article","type_order":7,"title":"Beyond the Spelling Miracle: Investigating Substring Awareness in Character-Blind Language Models","year":2025,"authors":["Ciaccio, C.","Sartor, M.","Miaschi, A.","Dell'Orletta, F."],"authors_source":"Ciaccio, C.; Sartor, M.; Miaschi, A.; Dell'Orletta, F.","authors_cnr_name":["Ciaccio, Cristiano","SARTOR, MARTA","MIASCHI, ALESSIO","DELL'ORLETTA, FELICE"],"authors_cnr_id":["rp27292","rp26663","rp12522","rp22811"],"authors_cnr_institute":[],"abstract":"Correctly identifying characters and substrings of words should be a basic but essential ability of any Language Model that aims to proficiently understand and produce language. Despite so, the majority of Pre-trained Language Models (PLMs) are \"character-blind\" and struggle in spelling tasks, although they still seem to acquire some character knowledge during pre-training, a phenomenon dubbed Spelling Miracle. To shed light on this phenomenon, we systematically evaluate a range of PLMs with different parameter sizes using a controlled binary substring identification task. Through a series of experiments, we propose the first comprehensive investigation on where, when, and how PLMs develop awareness of characters and substrings, with a particular linguistic focus on morphemic units such as prefixes, suffixes, and roots","keywords":["Large Language Models (LLMs)","Interpretability"],"pages":"11361-11372","url":"https:\/\/iris.cnr.it\/handle\/20.500.14243\/570461","volume":"","doi":"10.18653\/v1\/2025.findings-acl.593","editors":[],"editors_source":"","published":"Proceedings of the Annual Meeting of the Association for Computational Linguistics","publisher":"Association for Computational Linguistics (ACL)","issn":"","isbn":"","conference_name":"63rd Annual Meeting of the Association for Computational Linguistics, ACL 2025","conference_place":"","conference_date":"","last_updated_cnr":"2026-03-04 01:28:12","last_updated_oai":"2026-03-04 01:28:12","last_updated_www":"0000-00-00 00:00:00"},{"id":1364,"id_source":570444,"institutes":["ILC","IIT"],"type":"conference_article","type_order":7,"title":"Contextualized Counterspeech: Strategies for Adaptation, Personalization, and Evaluation","year":2025,"authors":["Cima, L.","Miaschi, A.","Trujillo, A.","Avvenuti, M.","Dell'Orletta, F.","Cresci, S."],"authors_source":"Cima, L.; Miaschi, A.; Trujillo, A.; Avvenuti, M.; Dell'Orletta, F.; Cresci, S.","authors_cnr_name":["CIMA, LORENZO","MIASCHI, ALESSIO","DELL'ORLETTA, FELICE","CRESCI, STEFANO"],"authors_cnr_id":["rp27644","rp12522","rp22811","rp05803"],"authors_cnr_institute":[],"abstract":"AI-generated counterspeech offers a promising and scalable strategy to curb online toxicity through direct replies that promote civil discourse. However, current counterspeech is one-size-fits-all, lacking adaptation to the moderation context and the users involved. We propose and evaluate multiple strategies for generating tailored counterspeech that is adapted to the moderation context and personalized for the moderated user. We instruct a LLaMA2-13B model to generate counterspeech, experimenting with various configurations based on different contextual information and fine-tuning strategies. We identify the configurations that generate persuasive counterspeech through a combination of quantitative indicators and human evaluations collected via a pre-registered mixed-design crowdsourcing experiment. Results show that contextualized counterspeech can significantly outperform state-of-the-art generic counterspeech in adequacy and persuasiveness, without compromising other characteristics. Our findings also reveal a poor correlation between quantitative indicators and human evaluations, suggesting that these methods assess different aspects and highlighting the need for nuanced evaluation methodologies. The effectiveness of contextualized AI-generated counterspeech and the divergence between human and algorithmic evaluations underscore the importance of increased human-AI collaboration in content moderation","keywords":["content moderation","Counterspeech","generative AI","online toxicity","personalization"],"pages":"5022-5033","url":"https:\/\/iris.cnr.it\/handle\/20.500.14243\/570444","volume":"","doi":"10.1145\/3696410.3714507","editors":[],"editors_source":"","published":"WWW 2025-Proceedings of the ACM Web Conference","publisher":"Association for Computing Machinery, Inc (1601 Broadway, 10th Floor, NEW YORK, NY, UNITED STATES)","issn":"","isbn":"","conference_name":"34th ACM Web Conference, WWW 2025","conference_place":"1601 Broadway, 10th Floor, NEW YORK, NY, UNITED STATES","conference_date":"","last_updated_cnr":"2026-03-04 01:27:58","last_updated_oai":"2026-03-04 01:27:58","last_updated_www":"0000-00-00 00:00:00"},{"id":2145,"id_source":552066,"institutes":["ILC","ISTI"],"type":"conference_article","type_order":7,"title":"Optimizing LLMs for Italian: reducing token fertility and enhancing efficiency through vocabulary adaptation","year":2025,"authors":["Moroni, L.","Puccetti, G.","Huguet Cabot, P. L.","Bejgu, A. S.","Barba, E.","Miaschi, A.","Dell'Orletta, F.","Esuli, A.","Navigli, R."],"authors_source":"Moroni, L.; Puccetti, G.; Huguet Cabot, P. -L.; Bejgu, A. S.; Barba, E.; Miaschi, A.; Dell'Orletta, F.; Esuli, A.; Navigli, R.","authors_cnr_name":["PUCCETTI, GIOVANNI","MIASCHI, ALESSIO","DELL'ORLETTA, FELICE","ESULI, ANDREA"],"authors_cnr_id":["rp13460","rp12522","rp22811","rp23515"],"authors_cnr_institute":[],"abstract":"The number of pretrained Large Language Models (LLMs) is increasing steadily, though the majority are designed predominantly for the English language. While state-of-the-art LLMs can handle other languages, due to language contamination or some degree of multilingual pretraining data, they are not optimized for non-English languages, leading to inefficient encoding (high token ``fertility'') and slower inference speed. In this work, we thoroughly compare a variety of vocabulary adaptation techniques for optimizing English LLMs for the Italian language, and put forward Semantic Alignment Vocabulary Adaptation (SAVA), a novel method that leverages neural mapping for vocabulary substitution. SAVA achieves competitive performance across multiple downstream tasks, enhancing grounded alignment strategies. We adapt two LLMs: Mistral-7B-v0. 1, reducing token fertility by 25{\\%}, and Llama-3. 1-8B, optimizing the vocabulary and reducing the number of parameters by 1 billion. We show that, following the adaptation of the vocabulary, these models can recover their performance with a relatively limited stage of continual training on the target language. Finally, we test the capabilities of the adapted models on various multi-choice and generative tasks","keywords":["Large Languiage Models, Italia LLM, Vocabulary Adaptation"],"pages":"6646-6660","url":"https:\/\/aclanthology.org\/2025.findings-naacl.371\/","volume":"","doi":"10.18653\/v1\/2025.findings-naacl.371","editors":[],"editors_source":"","published":"NAACL 2025 Findings proceedings","publisher":"Association for Computational Linguistics","issn":"","isbn":"979-8-89176-195-7","conference_name":"NAACL 2025-Annual Conference of the Nations of the Americas Chapter. Findings of the Association for Computational Linguistics","conference_place":"","conference_date":"","last_updated_cnr":"2026-04-26 07:06:07","last_updated_oai":"2026-04-26 07:06:07","last_updated_www":"0000-00-00 00:00:00"},{"id":105,"id_source":554367,"institutes":["ILC","ISTI"],"type":"conference_article","type_order":7,"title":"Stress-testing machine generated text detection: shifting language models writing style to fool detectors","year":2025,"authors":["Pedrotti, A.","Papucci, M.","Ciaccio, C.","Miaschi, A.","Puccetti, G.","Dell'Orletta, F.","Esuli, A."],"authors_source":"Pedrotti, A.; Papucci, M.; Ciaccio, C.; Miaschi, A.; Puccetti, G.; Dell'Orletta, F.; Esuli, A.","authors_cnr_name":["PEDROTTI, ANDREA","Papucci, Michele","Ciaccio, Cristiano","MIASCHI, ALESSIO","PUCCETTI, GIOVANNI","DELL'ORLETTA, FELICE","ESULI, ANDREA"],"authors_cnr_id":["rp13666","rp28269","rp27292","rp12522","rp13460","rp22811","rp23515"],"authors_cnr_institute":[],"abstract":"Recent advancements in Generative AI and Large Language Models (LLMs) have enabled the creation of highly realistic synthetic content, raising concerns about the potential for malicious use, such as misinformation and manipulation. Moreover, detecting Machine-Generated Text (MGT) remains challenging due to the lack of robust benchmarks that assess generalization to real-world scenarios. In this work, we evaluate the resilience of state-of-the-art MGT detectors (e. g., Mage, Radar, LLM-DetectAIve) to linguistically informed adversarial attacks. We develop a pipeline that fine-tunes language models using Direct Preference Optimization (DPO) to shift the MGT style toward human-written text (HWT), obtaining generations more challenging to detect by current models. Additionally, we analyze the linguistic shifts induced by the alignment and how detectors rely on \u201clinguistic shortcuts\u201d to detect texts. Our results show that detectors can be easily fooled with relatively few examples, resulting in a significant drop in detecting performances. This highlights the importance of improving detection methods and making them robust to unseen in-domain texts. We release code, models, and data to support future research on more robust MGT detection benchmarks","keywords":["machine-generated text detection, synthetic content detection"],"pages":"3010-3031","url":"https:\/\/aclanthology.org\/2025.findings-acl.156\/","volume":"","doi":"10.18653\/v1\/2025.findings-acl.156","editors":[],"editors_source":"","published":"NAACL 2025 Findings proceedings","publisher":"Association for Computational Linguistics","issn":"","isbn":"979-8-89176-256-5","conference_name":"NAACL 2025-Annual Conference of the Nations of the Americas Chapter of the Association for Computational Linguistics. Findings","conference_place":"","conference_date":"","last_updated_cnr":"2026-04-26 07:05:51","last_updated_oai":"2026-04-26 07:05:51","last_updated_www":"0000-00-00 00:00:00"},{"id":161,"id_source":570744,"institutes":["ILC"],"type":"conference_article","type_order":7,"title":"MAIA: a Benchmark for Multimodal AI Assessment","year":2025,"authors":["Testa, D.","Bonetta, G.","Bernardi, R.","Bondielli, A.","Lenci, A.","Miaschi, A.","Passaro, L.","Magnini, B."],"authors_source":"Testa, Davide; Bonetta, Giovanni; Bernardi, Raffaella; Bondielli, Alessandro; Lenci, Alessandro; Miaschi, Alessio; Passaro, Lucia; Magnini, Bernardo","authors_cnr_name":["MIASCHI, ALESSIO"],"authors_cnr_id":["rp12522"],"authors_cnr_institute":[],"abstract":"We introduce MAIA (Multimodal AI Assessment), a multimodal dataset developed as a core component of a competenceoriented benchmark designed for fine-grained investigation of the reasoning abilities of Visual Language Models (VLMs) on videos. The MAIA benchmark is characterized by several distinctive features. To the best of our knowledge, MAIA is the first Italian-native benchmark addressing video understanding: videos were carefully selected to reflect Italian culture, and the language data (ie, questions and reference answers) were produced by native-Italian speakers. Second, MAIA explicitly includes twelve reasoning categories that are specifically designed to assess the reasoning abilities of VLMs on videos. Third, we structured the dataset to support two aligned tasks (ie, a statement verification and an open-ended visual question answering) built on the same datapoints, this way allowing to assess VLM coherence across task formats. Finally MAIA integrates, by design, state-of-the-art LLMs in the development process of the benchmark, taking advantage of their linguistic and reasoning capabilities both for data augmentation and for assessing and improving the overall quality of the data. In the paper we focus on the design principles and the data collection methodology, highlighting how MAIA provides a significant advancement with respect to other available dataset for VLM benchmarking. Data available at GitHub","keywords":["multimodal, vllm, evaluation"],"pages":"","url":"https:\/\/iris.cnr.it\/handle\/20.500.14243\/570744","volume":"","doi":"","editors":[],"editors_source":"","published":"Proceedings of the Eleventh Italian Conference on Computational Linguistics (CLiC-it 2025)","publisher":"","issn":"","isbn":"","conference_name":"","conference_place":"","conference_date":"","last_updated_cnr":"2026-03-04 01:27:48","last_updated_oai":"2026-03-04 01:27:48","last_updated_www":"0000-00-00 00:00:00"},{"id":1209,"id_source":570743,"institutes":["ILC"],"type":"conference_article","type_order":7,"title":"All-in-one: Understanding and Generation in Multimodal Reasoning with the MAIA Benchmark","year":2025,"authors":["Testa, D.","Bonetta, G.","Bernardi, R.","Bondielli, A.","Lenci, A.","Miaschi, A.","Passaro, L.","Magnini, B."],"authors_source":"Testa, D.; Bonetta, G.; Bernardi, R.; Bondielli, A.; Lenci, A.; Miaschi, A.; Passaro, L.; Magnini, B.","authors_cnr_name":["LENCI, ALESSANDRA","MIASCHI, ALESSIO"],"authors_cnr_id":["rp10823","rp12522"],"authors_cnr_institute":[],"abstract":"We introduce MAIA (Multimodal AI Assessment), a native-Italian benchmark designed for fine-grained investigation of the reasoning abilities of visual language models on videos. MAIA differs from other available video benchmarks for its design, its reasoning categories, the metric it uses, and the language and culture of the videos. MAIA evaluates Vision Language Models (VLMs) on two aligned tasks: a visual statement verification task, and an openended visual question-answering task, both on the same set of video-related questions. It considers twelve reasoning categories that aim to disentangle language and vision relations by highlighting the role of the visual input. Thanks to its carefully taught design, it evaluates VLMs\u2019 consistency and visually grounded natural language comprehension and generation simultaneously through an aggregated metric revealing low results that highlight models\u2019 fragility. Last but not least, the video collection has been carefully selected to reflect the Italian culture, and the language data are produced by native-speakers. 1","keywords":["multimodal, vllm, multimodal reasoning"],"pages":"20030-20050","url":"https:\/\/iris.cnr.it\/handle\/20.500.14243\/570743","volume":"","doi":"10.18653\/v1\/2025.findings-emnlp.1091","editors":[],"editors_source":"","published":"Findings of the Association for Computational Linguistics: EMNLP 2025","publisher":"","issn":"","isbn":"","conference_name":"","conference_place":"","conference_date":"","last_updated_cnr":"2026-03-04 01:27:53","last_updated_oai":"2026-03-04 01:27:53","last_updated_www":"0000-00-00 00:00:00"},{"id":540,"id_source":518427,"institutes":["ILC"],"type":"conference_article","type_order":7,"title":"Evaluating Large Language Models via Linguistic Profiling","year":2024,"authors":["Miaschi, A.","Dell'Orletta, F.","Venturi, G."],"authors_source":"Miaschi, Alessio; Dell'Orletta, Felice; Venturi, Giulia","authors_cnr_name":["MIASCHI, ALESSIO","DELL'ORLETTA, FELICE","VENTURI, GIULIA"],"authors_cnr_id":["rp12522","rp22811","rp00732"],"authors_cnr_institute":[],"abstract":"Large Language Models (LLMs) undergo extensive evaluation against various benchmarks collected in established leaderboards to assess their performance across multiple tasks. However, to the best of our knowledge, there is a lack of comprehensive studies evaluating these models\u2019 linguistic abilities independent of specific tasks. In this paper, we introduce a novel evaluation methodology designed to test LLMs\u2019 sentence generation abilities under specific linguistic constraints. Drawing on the \u2018linguistic profiling\u2019 approach, we rigorously investigate the extent to which five LLMs of varying sizes, tested in both zero-and few-shot scenarios, effectively adhere to (morpho)syntactic constraints. Our findings shed light on the linguistic proficiency of LLMs, revealing both their capabilities and limitations in generating linguistically-constrained sentences","keywords":["Large Language Models, Controllable Text Generation, Linguistic Profiling"],"pages":"2835-2848","url":"https:\/\/aclanthology.org\/2024.emnlp-main.166","volume":"","doi":"10.18653\/v1\/2024.emnlp-main.166","editors":[],"editors_source":"","published":"Proceedings of the 2024 Conference on Empirical Methods in Natural Language Processing","publisher":"Association for Computational Linguistics (USA)","issn":"","isbn":"979-8-89176-164-3","conference_name":"Conference on Empirical Methods in Natural Language Processing (EMNLP)","conference_place":"USA","conference_date":"","last_updated_cnr":"2025-02-08 06:50:29","last_updated_oai":"2025-02-08 06:50:29","last_updated_www":"0000-00-00 00:00:00"},{"id":437,"id_source":487005,"institutes":["ILC"],"type":"conference_article","type_order":7,"title":"Linguistic Knowledge Can Enhance Encoder-Decoder Models (If You Let It)","year":2024,"authors":["Miaschi, A.","Dell'Orletta, F.","Venturi, G."],"authors_source":"Miaschi, Alessio; Dell'Orletta, Felice; Venturi, Giulia","authors_cnr_name":["MIASCHI, ALESSIO","DELL'ORLETTA, FELICE","VENTURI, GIULIA"],"authors_cnr_id":["rp12522","rp22811","rp00732"],"authors_cnr_institute":[],"abstract":"In this paper, we explore the impact of augmenting pre-trained Encoder-Decoder models, specifically T5, with linguistic knowledge for the prediction of a target task. In particular, we investigate whether fine-tuning a T5 model on an intermediate task that predicts structural linguistic properties of sentences modifies its performance in the target task of predicting sentence-level complexity. Our study encompasses diverse experiments conducted on Italian and English datasets, employing both monolingual and multilingual T5 models at various sizes. Results obtained for both languages and in cross-lingual configurations show that linguistically motivated intermediate fine-tuning has generally a positive impact on target task performance, especially when applied to smaller models and in scenarios with limited data availability","keywords":["encoder-decoder, intermediate fine-tuning, linguistic features, sentence complexity"],"pages":"10539-10554","url":"https:\/\/aclanthology.org\/2024.lrec-main.922\/","volume":"","doi":"","editors":[],"editors_source":"","published":"Proceedings of the 2024 Joint International Conference on Computational Linguistics, Language Resources and Evaluation (LREC-COLING 2024)","publisher":"ELRA and ICCL","issn":"","isbn":"978-2-493814-10-4","conference_name":"Joint International Conference on Computational Linguistics, Language Resources and Evaluation (LREC-COLING 2024)","conference_place":"","conference_date":"","last_updated_cnr":"2025-03-05 05:11:23","last_updated_oai":"2025-03-05 05:11:23","last_updated_www":"0000-00-00 00:00:00"},{"id":1072,"id_source":519997,"institutes":["ILC"],"type":"conference_article","type_order":7,"title":"T-FREX: A Transformer-based Feature Extraction Method from Mobile App Reviews","year":2024,"authors":["Motger, Q.","Miaschi, A.","Dell'Orletta, F.","Franch, X.","Marco, J."],"authors_source":"Motger, Q.; Miaschi, A.; Dell'Orletta, F.; Franch, X.; Marco, J.","authors_cnr_name":["MIASCHI, ALESSIO","DELL'ORLETTA, FELICE"],"authors_cnr_id":["rp12522","rp22811"],"authors_cnr_institute":[],"abstract":"Mobile app reviews are a large-scale data source for software-related knowledge generation activities, including software maintenance, evolution and feedback analysis. Effective extraction of features (i. e., functionalities or characteristics) from these reviews is key to support analysis on the acceptance of these features, identification of relevant new feature requests and prioritization of feature development, among others. Traditional methods focus on syntactic pattern-based approaches, typically context-agnostic, evaluated on a closed set of apps, difficult to replicate and limited to a reduced set and domain of apps. Mean-while, the pervasiveness of Large Language Models (LLMs) based on the Transformer architecture in software engineering tasks lays the groundwork for empirical evaluation of the performance of these models to support feature extraction. In this study, we present T-FREX, a Transformer-based, fully automatic approach for mobile app review feature extraction. First, we collect a set of ground truth features from users in a real crowdsourced software recommendation platform and transfer them automatically into a dataset of app reviews. Then, we use this newly created dataset to fine-tune multiple LLMs on a named entity recognition task under different data configurations. We assess the performance of T-FREX with respect to this ground truth, and we complement our analysis by comparing T-FREX with a baseline method from the field. Finally, we assess the quality of new features predicted by T-FREX through an external human evaluation. Results show that T-FREX outperforms on average the traditional syntactic-based method, especially when discovering new features from a domain for which the model has been fine-tuned","keywords":["feature extraction","large language models","mobile apps","named entity recognition","reviews","token classification"],"pages":"227-238","url":"https:\/\/iris.cnr.it\/handle\/20.500.14243\/519997","volume":"","doi":"10.1109\/SANER60148.2024.00030","editors":[],"editors_source":"","published":"Proceedings-2024 IEEE International Conference on Software Analysis, Evolution and Reengineering, SANER 2024","publisher":"Institute of Electrical and Electronics Engineers Inc","issn":"","isbn":"","conference_name":"31st IEEE International Conference on Software Analysis, Evolution and Reengineering, SANER 2024","conference_place":"","conference_date":"","last_updated_cnr":"2025-06-14 00:28:52","last_updated_oai":"2025-06-14 00:28:52","last_updated_www":"0000-00-00 00:00:00"},{"id":1435,"id_source":439017,"institutes":["ILC"],"type":"journal_article","type_order":1,"title":"Tell me how you write and I'll tell you what you read: a study on the writing style of book reviews","year":2023,"authors":["Alzetta, C.","Dell'Orletta, F.","Miaschi, A.","Prat, E.","Venturi, G."],"authors_source":"Alzetta, Chiara; Dell'Orletta, Felice; Miaschi, Alessio; Prat, Elena; Venturi, Giulia","authors_cnr_name":["ALZETTA, CHIARA","DELL'ORLETTA, FELICE","MIASCHI, ALESSIO","VENTURI, GIULIA"],"authors_cnr_id":["rp12530","rp22811","rp12522","rp00732"],"authors_cnr_institute":[],"abstract":"The paper aims at investigating variations in the writing style of book reviews published on different social reading platforms and referring to books of different genres, which enables acquiring insights into communication strategies adopted by readers to share their reading experiences. To this end, we introduce a corpus-based study focused on the analysis of A Good Review, a novel corpus of online book reviews written in Italian, posted on Amazon and Goodreads, and covering six literary fiction genres. We rely on stylometric analysis to explore the linguistic properties and lexicon of reviews and the authors conducted automatic classification experiments using multiple approaches and feature configurations to predict either the review's platform or the literary genre. The analysis of user-generated reviews demonstrates that language is a quite variable dimension across reading platforms, but not as much across book genres. The classification experiments revealed that features modelling the syntactic structure of the sentence are reliable proxies for discerning Amazon and Goodreads reviews, whereas lexical information showed a higher predictive role for automatically discriminating the genre","keywords":["Stylometric analysis","Textual Genre detection","Book reviews"],"pages":"23","url":"https:\/\/www.emerald.com\/insight\/content\/doi\/10.1108\/JD-04-2023-0073\/full\/html","volume":"79","doi":"10.1108\/JD-04-2023-0073","editors":[],"editors_source":"","published":"JOURNAL OF DOCUMENTATION","publisher":"","issn":"0022-0418","isbn":"","conference_name":"","conference_place":"","conference_date":"","last_updated_cnr":"2025-03-06 04:29:16","last_updated_oai":"2025-03-06 04:29:16","last_updated_www":"0000-00-00 00:00:00"},{"id":2064,"id_source":439018,"institutes":["ILC"],"type":"journal_article","type_order":1,"title":"Testing the Effectiveness of the Diagnostic Probing Paradigm on Italian Treebanks","year":2023,"authors":["Miaschi, A.","Alzetta, C.","Brunato, D.","Dell'Orletta, F.","Venturi, G."],"authors_source":"Miaschi, Alessio; Alzetta, Chiara; Brunato, Dominique; Dell'Orletta, Felice; Venturi, Giulia","authors_cnr_name":["MIASCHI, ALESSIO","ALZETTA, CHIARA","BRUNATO, DOMINIQUE PIERINA","DELL'ORLETTA, FELICE","VENTURI, GIULIA"],"authors_cnr_id":["rp12522","rp12530","rp06836","rp22811","rp00732"],"authors_cnr_institute":[],"abstract":"The outstanding performance recently reached by neural language models (NLMs) across many natural language processing (NLP) tasks has steered the debate towards understanding whether NLMs implicitly learn linguistic competence. Probes, i. e., supervised models trained using NLM representations to predict linguistic properties, are frequently adopted to investigate this issue. However, it is still questioned if probing classification tasks really enable such investigation or if they simply hint at surface patterns in the data. This work contributes to this debate by presenting an approach to assessing the effectiveness of a suite of probing tasks aimed at testing the linguistic knowledge implicitly encoded by one of the most prominent NLMs, BERT. To this aim, we compared the performance of probes when predicting gold and automatically altered values of a set of linguistic features. Our experiments were performed on Italian and were evaluated across BERT's layers and for sentences with different lengths. As a general result, we observed higher performance in the prediction of gold values, thus suggesting that the probing model is sensitive to the distortion of feature values. However, our experiments also showed that the length of a sentence is a highly influential factor that is able to confound the probing model's predictions","keywords":["Neural language model","Probing tasks","Treebanks"],"pages":"19","url":"https:\/\/www.mdpi.com\/2078-2489\/14\/3\/144","volume":"14 (3)","doi":"10.3390\/info14030144","editors":[],"editors_source":"","published":"INFORMATION","publisher":"","issn":"2078-2489","isbn":"","conference_name":"","conference_place":"","conference_date":"","last_updated_cnr":"2025-03-14 22:18:11","last_updated_oai":"2025-03-14 22:18:11","last_updated_www":"0000-00-00 00:00:00"},{"id":1544,"id_source":470901,"institutes":["ILC"],"type":"conference_article","type_order":7,"title":"LangLearn at EVALITA 2023: Overview of the Language Learning Development Task","year":2023,"authors":["Alzetta, C.","Brunato, D.","Dell'Orletta, F.","Miaschi, A.","Sagae, K.","S\u00e1nchez Guti\u00e9rrez, C. H.","Venturi, G."],"authors_source":"Alzetta, Chiara; Brunato, Dominique; Dell'Orletta, Felice; Miaschi, Alessio; Sagae, Kenji; S\u00e1nchez-Guti\u00e9rrez, Claudia H.; Venturi, Giulia","authors_cnr_name":["ALZETTA, CHIARA","BRUNATO, DOMINIQUE PIERINA","DELL'ORLETTA, FELICE","MIASCHI, ALESSIO","VENTURI, GIULIA"],"authors_cnr_id":["rp12530","rp06836","rp22811","rp12522","rp00732"],"authors_cnr_institute":[],"abstract":"Language Learning Development (LangLearn) is the EVALITA 2023 shared task on automatic language development assessment, which consists in predicting the evolution of the written language abilities of learners across time. LangLearn is conceived to be multilingual, relying on written productions of Italian and Spanish learners, and representative of L1 and L2 learning scenarios. A total of 9 systems were submitted by 5 teams. The results highlight the open challenges of automatic language development assessment","keywords":[],"pages":"","url":"https:\/\/iris.cnr.it\/handle\/20.500.14243\/470901","volume":"","doi":"","editors":[],"editors_source":"","published":"Proceedings of EVALITA 2023","publisher":"Accademia University Press (Torino, ITA)","issn":"","isbn":"9791255000693","conference_name":"8th Evaluation Campaign of Natural Language Processing and Speech Tools for Italian","conference_place":"Torino","conference_date":"","last_updated_cnr":"2025-01-24 23:18:36","last_updated_oai":"2025-01-24 23:18:36","last_updated_www":"0000-00-00 00:00:00"},{"id":1525,"id_source":470921,"institutes":["ILC"],"type":"conference_article","type_order":7,"title":"Unmasking the Wordsmith: Revealing Author Identity through Reader Reviews","year":2023,"authors":["Alzetta, C.","Dell'Orletta, F.","Fazzone, C.","Miaschi, A.","Venturi, G."],"authors_source":"Alzetta, Chiara; Dell'Orletta, Felice; Fazzone, Chiara; Miaschi, Alessio; Venturi, Giulia","authors_cnr_name":["ALZETTA, CHIARA","DELL'ORLETTA, FELICE","FAZZONE, CHIARA","MIASCHI, ALESSIO","VENTURI, GIULIA"],"authors_cnr_id":["rp12530","rp22811","rp16373","rp12522","rp00732"],"authors_cnr_institute":[],"abstract":"Traditional genre-based approaches for book recommendations face challenges due to the vague definition of genres. To overcome this, we propose a novel task called Book Author Prediction, where we predict the author of a book based on user-generated reviews\u2019 writing style. To this aim, we first introduce the \u2018Literary Voices Corpus\u2019 (LVC), a dataset of Italian book reviews, and use it to train and test machine learning models. Our study contributes valuable insights for developing user-centric systems that recommend leisure readings based on individual readers\u2019 interests and writing styles","keywords":[],"pages":"","url":"https:\/\/ceur-ws.org\/Vol-3596\/paper4.pdf","volume":"","doi":"","editors":[],"editors_source":"","published":"Proceedings of the 9th Italian Conference on Computational Linguistics","publisher":"","issn":"","isbn":"","conference_name":"9th Italian Conference on Computational Linguistics","conference_place":"","conference_date":"","last_updated_cnr":"2025-01-24 23:18:13","last_updated_oai":"2025-01-24 23:18:13","last_updated_www":"0000-00-00 00:00:00"},{"id":136,"id_source":520527,"institutes":["ILC"],"type":"conference_article","type_order":7,"title":"Lost in Labels: An Ongoing Quest to Optimize Text-to-Text Label Selection for Classification","year":2023,"authors":["Miaschi, A.","Papucci, M.","Dell'Orletta, F."],"authors_source":"Miaschi, Alessio; Papucci, Michele; Dell'Orletta, Felice","authors_cnr_name":["MIASCHI, ALESSIO","Papucci, Michele","DELL'ORLETTA, FELICE"],"authors_cnr_id":["rp12522","rp28269","rp22811"],"authors_cnr_institute":[],"abstract":"In this paper, we present an evaluation of the influence of label selection on the performance of a Sequence-to-Sequence Transformer model in a classification task. Our study investigates whether the choice of words used to represent classification categories affects the model\u2019s performance, and if there exists a relationship between the model\u2019s performance and the selected words. To achieve this, we fine-tuned an Italian T5 model on topic classification using various labels. Our results indicate that the different label choices can significantly impact the model\u2019s performance. That being said, we did not find a clear answer on how these choices affect the model performances, highlighting the need for further research in optimizing label selection","keywords":["encoder-decoder, label selection, topic classification"],"pages":"","url":"https:\/\/iris.cnr.it\/handle\/20.500.14243\/520527","volume":"516 (394)","doi":"","editors":[],"editors_source":"","published":"Proceedings of the 9th Italian Conference on Computational Linguistics CLiC-it 2023: Venice, Italy, November 30-December 2, 2023","publisher":"","issn":"","isbn":"","conference_name":"","conference_place":"","conference_date":"","last_updated_cnr":"2024-12-22 02:46:01","last_updated_oai":"2024-12-22 02:46:01","last_updated_www":"0000-00-00 00:00:00"},{"id":1383,"id_source":417257,"institutes":["ILC"],"type":"journal_article","type_order":1,"title":"On Robustness and Sensitivity of a Neural Language Model: A Case Study on Italian L1 Learner Errors","year":2022,"authors":["Miaschi, A.","Brunato, D. P.","Dell'Orletta, F.","Venturi, G."],"authors_source":"Miaschi, Alessio; Alessio, ; Brunato, DOMINIQUE PIERINA; Dominique, ; Dell'Orletta, Felice; Felice, ; Venturi, Giulia; Giulia,","authors_cnr_name":["MIASCHI, ALESSIO","BRUNATO, DOMINIQUE PIERINA","DELL'ORLETTA, FELICE","VENTURI, GIULIA"],"authors_cnr_id":["rp12522","rp06836","rp22811","rp00732"],"authors_cnr_institute":[],"abstract":"In this paper, we propose a comprehensive linguistic study aimed at assessing the implicit behavior of one of the most prominent Neural Language Models (NLM) based on Transformer architectures, BERT (Devlin et al., 2019), when dealing with a particular source of noisy data, namely essays written by L1 Italian learners containing a variety of errors targeting grammar, orthography and lexicon. Differently from previous works, we focus on the pre-training stage and we devise two complementary evaluation tasks aimed at assessing the impact of errors on sentence-level inner representations in terms of semantic robustness and linguistic sensitivity. While the first evaluation perspective is meant to probe the model's ability to encode the semantic similarity between sentences also in the presence of errors, the second type of probing task evaluates the influence of errors on BERT's implicit knowledge of a set of raw and morpho-syntactic properties of a sentence. Our experiments show that BERT's ability to compute sentence similarity and to correctly encode multi-leveled linguistic information of a sentence are differently modulated by the category of errors and that the error hierarchies in terms of robustness and sensitivity change across layer-wise representations","keywords":["Natural Language Processing","Neural Language Model","Interpretability"],"pages":"426-438","url":"https:\/\/doi.org\/10.1109\/TASLP.2022.3226333","volume":"31","doi":"10.1109\/TASLP.2022.3226333","editors":[],"editors_source":"","published":"IEEE\/ACM TRANSACTIONS ON AUDIO, SPEECH, AND LANGUAGE PROCESSING","publisher":"","issn":"2329-9290","isbn":"","conference_name":"","conference_place":"","conference_date":"","last_updated_cnr":"2025-05-07 00:38:54","last_updated_oai":"2025-05-07 00:38:54","last_updated_www":"0000-00-00 00:00:00"},{"id":1747,"id_source":443057,"institutes":["ILC"],"type":"journal_article","type_order":1,"title":"Probing Linguistic Knowledge in Italian Neural Language Models across Language Varieties","year":2022,"authors":["Miaschi, A.","Sarti, G.","Brunato, D. P.","Dell'Orletta, F.","Venturi, G."],"authors_source":"Miaschi, Alessio; Miaschi, Alessio; Sarti, ; Gabriele, ; Brunato, DOMINIQUE PIERINA; Brunato, DOMINIQUE PIERINA; Dell'Orletta, Felice; Dell'Orletta, Felice; Venturi, Giulia; Venturi, Giulia","authors_cnr_name":["MIASCHI, ALESSIO","MIASCHI, ALESSIO","BRUNATO, DOMINIQUE PIERINA","BRUNATO, DOMINIQUE PIERINA","DELL'ORLETTA, FELICE","DELL'ORLETTA, FELICE","VENTURI, GIULIA","VENTURI, GIULIA"],"authors_cnr_id":["rp12522","rp12522","rp06836","rp06836","rp22811","rp22811","rp00732","rp00732"],"authors_cnr_institute":[],"abstract":"In this paper, we present an in-depth investigation of the linguistic knowledge encoded by the transformer models currently available for the Italian language. In particular, we investigate how the complexity of two different architectures of probing models affects the performance of the Transformers in encoding a wide spectrum of linguistic features. Moreover, we explore how this implicit knowledge varies according to different textual genres and language varieties","keywords":["Neural Language Models","Interpretability","Language Varieties"],"pages":"25-44","url":"http:\/\/www.aaccademia.it\/ita\/scheda-libro?aaref=1518","volume":"","doi":"10.4000\/ijcol.965","editors":[],"editors_source":"","published":"IJCOL","publisher":"","issn":"2499-4553","isbn":"","conference_name":"","conference_place":"","conference_date":"","last_updated_cnr":"2025-02-05 22:52:50","last_updated_oai":"2025-02-05 22:52:50","last_updated_www":"0000-00-00 00:00:00"},{"id":1250,"id_source":443056,"institutes":["ILC"],"type":"conference_article","type_order":7,"title":"Punctuation Restoration in\u00a0Spoken Italian Transcripts with\u00a0Transformers","year":2022,"authors":["Miaschi, A.","Ravelli, A.","Dell'Orletta, F."],"authors_source":"Miaschi, A; Ravelli, Aa; Dell'Orletta, F","authors_cnr_name":["MIASCHI, ALESSIO","RAVELLI, ANDREA AMELIO","DELL'ORLETTA, FELICE"],"authors_cnr_id":["rp12522","rp13673","rp22811"],"authors_cnr_institute":[],"abstract":"In this paper, we propose an evaluation of a Transformer-based punctuation restoration model for the Italian language. Experimenting with a BERT-base model, we perform several fine-tuning with different training data and sizes and tested them in an in-and cross-domain scenario. Moreover, we conducted an error analysis of the main weaknesses of the model related to specific punctuation marks. Finally, we test our system either quantitatively and qualitatively, by offering a typical task-oriented and a perception-based acceptability evaluation","keywords":["nlp","transformer models","puncutation restoration"],"pages":"245-260","url":"http:\/\/www.scopus.com\/record\/display.url?eid=2-s2.0-85135083576&origin=inward","volume":"13196 LNAI","doi":"10.1007\/978-3-031-08421-8_17","editors":[],"editors_source":"","published":"Proccedings of AIxIA 2021-Advances in Artificial Intelligence","publisher":"","issn":"","isbn":"","conference_name":"AIxIA 2021-Advances in Artificial Intelligence","conference_place":"","conference_date":"","last_updated_cnr":"2024-12-24 16:10:00","last_updated_oai":"2024-12-24 16:10:00","last_updated_www":"0000-00-00 00:00:00"},{"id":1621,"id_source":415084,"institutes":["ILC"],"type":"conference_article","type_order":7,"title":"Evaluating Text-To-Text Framework for Topic and Style Classification of Italian texts","year":2022,"authors":["Papucci, M.","De Nigris, C.","Miaschi, A.","Dell'Orletta, F."],"authors_source":"Papucci, Michele; De Nigris, Chiara; Miaschi, Alessio; Dell'Orletta, Felice","authors_cnr_name":["MIASCHI, ALESSIO","DELL'ORLETTA, FELICE"],"authors_cnr_id":["rp12522","rp22811"],"authors_cnr_institute":[],"abstract":"In this paper, we propose an extensive evaluation of the first text-to-text Italian Neural Language Model (NLM), IT5 [1], on a classification scenario. In particular, we test the performance of IT5 on several tasks involving both the classification of the topic and the style of a set of Italian posts. We assess the model in two different configurations, single-and multi-task classification, and we compare it with a more traditional NLM based on the Transformer architecture (i. e. BERT). Moreover, we test its performance in a few-shot learning scenario. We also perform a qualitative investigation on the impact of label representations in modeling the classification of the IT5 model. Results show that IT5 could achieve good results, although generally lower than the BERT model. Nevertheless, we observe a significant performance improvement of the Text-to-text model in a multi-task classification scenario. Finally, we found that altering the representation of the labels mainly impacts the classification of the topic","keywords":["bert","style classification","t5","text-to-text","topic classification","transformers"],"pages":"56-70","url":"http:\/\/www.scopus.com\/record\/display.url?eid=2-s2.0-85143252156&origin=inward","volume":"3287","doi":"","editors":[],"editors_source":"","published":"","publisher":"","issn":"","isbn":"","conference_name":"Sixth Workshop on Natural Language for Artificial Intelligence, NL4AI 2022","conference_place":"","conference_date":"","last_updated_cnr":"2025-06-15 00:25:28","last_updated_oai":"2025-06-15 00:25:28","last_updated_www":"0000-00-00 00:00:00"},{"id":945,"id_source":402654,"institutes":["ILC"],"type":"journal_article","type_order":1,"title":"A NLP-based stylometric approach for tracking the evolution of L1 written language competence","year":2021,"authors":["Miaschi, A.","Brunato, D. P.","Dell'Orletta, F."],"authors_source":"Miaschi, Alessio; Miaschi, Alessio; Brunato, DOMINIQUE PIERINA; Brunato, DOMINIQUE PIERINA; Dell'Orletta, Felice; Dell'Orletta, Felice","authors_cnr_name":["MIASCHI, ALESSIO","MIASCHI, ALESSIO","BRUNATO, DOMINIQUE PIERINA","BRUNATO, DOMINIQUE PIERINA","DELL'ORLETTA, FELICE","DELL'ORLETTA, FELICE"],"authors_cnr_id":["rp12522","rp12522","rp06836","rp06836","rp22811","rp22811"],"authors_cnr_institute":[],"abstract":"In this study we present a Natural Language Processing (NLP)-based stylometric approach for tracking the evolution of written language competence in Italian L1 learners. The approach relies on a wide set of linguistically motivated features capturing stylistic aspects of a text, which were extracted from students' essays contained in CItA (Corpus Italiano di Apprendenti L1), the first longitudinal corpus of texts written by Italian L1 learners enrolled in the first and second year of lower secondary school. We address the problem of modeling written language development as a supervised classification task consisting in predicting the chronological order of essays written by the same student at different temporal spans. The promising results obtained in several classification scenarios allow us to conclude that it is possible to automatically model the highly relevant changes affecting written language evolution across time, as well as identifying which features are more predictive of this process. In the last part of the article, we focus the attention on the possible influence of background variables on language learning and we present preliminary results of a pilot study aiming at understanding how the observed developmental patterns are affected by information related to the school environment of the student","keywords":["stylometry","computational linguistics","language competence"],"pages":"71-105","url":"https:\/\/www.jowr.org\/abstracts\/vol13_1\/Miaschi_et_al_2021_13_1_abstract.html","volume":"VOL. 13","doi":"10.17239\/jowr-2021.13.01.03","editors":[],"editors_source":"","published":"JOURNAL OF WRITING RESEARCH","publisher":"","issn":"2030-1006","isbn":"","conference_name":"","conference_place":"","conference_date":"","last_updated_cnr":"2025-03-01 07:05:12","last_updated_oai":"2025-03-01 07:05:12","last_updated_www":"0000-00-00 00:00:00"},{"id":1529,"id_source":440996,"institutes":["ILC"],"type":"conference_article","type_order":7,"title":"On the role of textual connectives in sentence comprehension: A new dataset for Italian","year":2021,"authors":["Albertin, G.","Miaschi, A.","Brunato, D."],"authors_source":"Albertin G.; Miaschi A.; Brunato D.","authors_cnr_name":["MIASCHI, ALESSIO","BRUNATO, DOMINIQUE PIERINA"],"authors_cnr_id":["rp12522","rp06836"],"authors_cnr_institute":[],"abstract":"In this paper we present a new evaluation resource for Italian aimed at assessing the role of textual connectives in the comprehension of the meaning of a sentence. The resource is arranged in two sections (acceptability assessment and cloze test), each one corresponding to a distinct challenge task conceived to test how subtle modifications involving connectives in real usage sentences influence the perceived acceptability of the sentence by native speakers and Neural Language Models (NLMs). Although the main focus is the presentation of the dataset, we also provide some preliminary data comparing human judgments and NLMs performance in the two tasks","keywords":["neural language models","textual connectives","sentence acceptability"],"pages":"","url":"http:\/\/ceur-ws.org\/Vol-3033\/paper16.pdf","volume":"3033","doi":"","editors":[],"editors_source":"","published":"","publisher":"","issn":"","isbn":"","conference_name":"8th Italian Conference on Computational Linguistics (CLIC-it 2021)","conference_place":"","conference_date":"","last_updated_cnr":"2025-06-14 00:42:45","last_updated_oai":"2025-06-14 00:42:45","last_updated_www":"0000-00-00 00:00:00"},{"id":1396,"id_source":400472,"institutes":["ILC"],"type":"conference_article","type_order":7,"title":"A dissemination workshop for introducing young Italian students to NLP","year":2021,"authors":["Messina, L.","Busso, L.","Combei, C. R.","Miaschi, A.","Pannitto, L.","Sarti, G.","Nissim, M."],"authors_source":"Messina, ; Lucio, ; Busso, ; Lucia, ; Combei, ; Claudia, Roberta; Miaschi, Alessio; Miaschi, Alessio; Pannitto, ; Ludovica, ; Sarti, ; Gabriele, ; Nissim, ; Malvina,","authors_cnr_name":["MIASCHI, ALESSIO","MIASCHI, ALESSIO"],"authors_cnr_id":["rp12522","rp12522"],"authors_cnr_institute":[],"abstract":"We describe and make available the game-based material developed for a laboratory run at several Italian science festivals to popularize NLP among young students","keywords":["nlp","teaching"],"pages":"52-54","url":"https:\/\/www.aclweb.org\/anthology\/2021.teachingnlp-1.7","volume":"","doi":"","editors":[],"editors_source":"","published":"Proceedings of the 5th Workshop on Teaching NLP","publisher":"","issn":"","isbn":"978-1-954085-36-7","conference_name":"5th Workshop on Teaching NLP","conference_place":"","conference_date":"","last_updated_cnr":"2024-12-22 03:04:22","last_updated_oai":"2024-12-22 03:04:22","last_updated_www":"0000-00-00 00:00:00"},{"id":881,"id_source":446048,"institutes":["ILC"],"type":"conference_article","type_order":7,"title":"Probing tasks under pressure","year":2021,"authors":["Miaschi, A.","Alzetta, C.","Brunato, D.","Dell'Orletta, F.","Venturi, G."],"authors_source":"Miaschi A.; Alzetta C.; Brunato D.; Dell'Orletta F.; Venturi G.","authors_cnr_name":["MIASCHI, ALESSIO","ALZETTA, CHIARA","BRUNATO, DOMINIQUE PIERINA","DELL'ORLETTA, FELICE","VENTURI, GIULIA"],"authors_cnr_id":["rp12522","rp12530","rp06836","rp22811","rp00732"],"authors_cnr_institute":[],"abstract":"Probing tasks are frequently used to evaluate whether the representations of Neural Language Models (NLMs) encode linguistic information. However, it is still questioned if probing classification tasks really enable such investigation or they simply hint for surface patterns in the data. We present a method to investigate this question by comparing the accuracies of a set of probing tasks on gold and automatically generated control datasets. Our results suggest that probing tasks can be used as reliable diagnostic methods to investigate the linguistic information encoded in NLMs representations","keywords":["Neural Language Models","Linguistic probing","Treebanks"],"pages":"1-7","url":"http:\/\/ceur-ws.org\/Vol-3033\/paper29.pdf","volume":"3033","doi":"","editors":[],"editors_source":"","published":"","publisher":"","issn":"","isbn":"","conference_name":"8th Italian Conference on Computational Linguistics (CLIC-it 2021)","conference_place":"","conference_date":"","last_updated_cnr":"2024-12-22 03:17:12","last_updated_oai":"2024-12-22 03:17:12","last_updated_www":"0000-00-00 00:00:00"},{"id":1549,"id_source":400474,"institutes":["ILC"],"type":"conference_article","type_order":7,"title":"What Makes My Model Perplexed? A Linguistic Investigation on Neural Language Models Perplexity","year":2021,"authors":["Miaschi, A.","Brunato, D. P.","Dell'Orletta, F.","Venturi, G."],"authors_source":"Miaschi, Alessio; Miaschi, Alessio; Brunato, DOMINIQUE PIERINA; Brunato, DOMINIQUE PIERINA; Dell'Orletta, Felice; Dell'Orletta, Felice; Venturi, Giulia; Venturi, Giulia","authors_cnr_name":["MIASCHI, ALESSIO","MIASCHI, ALESSIO","BRUNATO, DOMINIQUE PIERINA","BRUNATO, DOMINIQUE PIERINA","DELL'ORLETTA, FELICE","DELL'ORLETTA, FELICE","VENTURI, GIULIA","VENTURI, GIULIA"],"authors_cnr_id":["rp12522","rp12522","rp06836","rp06836","rp22811","rp22811","rp00732","rp00732"],"authors_cnr_institute":[],"abstract":"This paper presents an investigation aimed at studying how the linguistic structure of a sentence affects the perplexity of two of the most popular Neural Language Models (NLMs), BERT and GPT-2. We first compare the sentence-level likelihood computed with BERT and the GPT-2's perplexity showing that the two metrics are correlated. In addition, we exploit linguistic features capturing a wide set of morpho-syntactic and syntactic phenomena showing how they contribute to predict the perplexity of the two NLMs","keywords":["nlp","interpretability","deep learning"],"pages":"40-47","url":"https:\/\/www.aclweb.org\/anthology\/2021.deelio-1.5","volume":"","doi":"","editors":[],"editors_source":"","published":"Proceedings of the 2nd Workshop on Knowledge Extraction and Integrationfor Deep Learning Architectures","publisher":"","issn":"","isbn":"978-1-954085-30-5","conference_name":"2nd Workshop on Knowledge Extraction and Integrationfor Deep Learning Architectures","conference_place":"","conference_date":"","last_updated_cnr":"2024-12-20 02:05:59","last_updated_oai":"2024-12-20 02:05:59","last_updated_www":"0000-00-00 00:00:00"},{"id":1974,"id_source":443055,"institutes":["ILC"],"type":"conference_article","type_order":7,"title":"Evaluating Transformer Models for Punctuation Restoration in Italian","year":2021,"authors":["Miaschi, A.","Ravelli, A. A.","Dell'Orletta, F."],"authors_source":"Miaschi A.; Ravelli A.A.; Dell'Orletta F.","authors_cnr_name":["MIASCHI, ALESSIO","RAVELLI, ANDREA AMELIO","DELL'ORLETTA, FELICE"],"authors_cnr_id":["rp12522","rp13673","rp22811"],"authors_cnr_institute":[],"abstract":"In this paper, we propose an evaluation of a Transformerbased punctuation restoration model for the Italian language. Experimenting with a BERT-base model, we perform several fine-tuning with different training data and sizes and tested them in an in-and crossdomain scenario. Moreover, we offer a comparison in a multilingual setting with the same model fine-tuned on English transcriptions. Finally, we conclude with an error analysis of the main weaknesses of the model related to specific punctuation marks","keywords":["transformer models","nlp","punctuation restoration"],"pages":"","url":"http:\/\/www.scopus.com\/record\/display.url?eid=2-s2.0-85121647978&origin=inward","volume":"3015","doi":"","editors":[],"editors_source":"","published":"","publisher":"","issn":"","isbn":"","conference_name":"5th Workshop on Natural Language for Artificial Intelligence (NL4AI 2021)","conference_place":"","conference_date":"","last_updated_cnr":"2025-06-15 00:19:20","last_updated_oai":"2025-06-15 00:19:20","last_updated_www":"0000-00-00 00:00:00"},{"id":714,"id_source":400471,"institutes":["ILC"],"type":"conference_article","type_order":7,"title":"Teaching NLP with Bracelets and Restaurant Menus: An Interactive Workshop for Italian Students","year":2021,"authors":["Pannitto, L.","Busso, L.","Combei, C. R.","Messina, L.","Miaschi, A.","Sarti, G.","Nissim, M."],"authors_source":"Pannitto, ; and Busso, Ludovica; and Combei, Lucia; Roberta and Messina, Claudia; and Miaschi, Lucio; and Sarti, Alessio; and Nissim, Gabriele; Ludovica Pannitto, Malvina; Busso, Lucia; Roberta Combei, Claudia; Messina, Lucio; Miaschi, Alessio; Sarti, Gabriele; Nissim, Malvina","authors_cnr_name":["MIASCHI, ALESSIO"],"authors_cnr_id":["rp12522"],"authors_cnr_institute":[],"abstract":"Although Natural Language Processing is at the core of many tools young people use in their everyday life, high school curricula (in Italy) do not include any computational linguistics education. This lack of exposure makes the use of such tools less responsible than it could be, and makes choosing computational linguistics as a university degree unlikely. To raise awareness, curiosity, and longer-term interest in young people, we have developed an interactive workshop designed to illustrate the basic principles of NLP and computational linguistics to high school Italian students aged between 13 and 18 years. The workshop takes the form of a game in which participants play the role of machines needing to solve some of the most common problems a computer faces in understanding language: from voice recognition to Markov chains to syntactic parsing. Participants are guided through the workshop with the help of instructors, who present the activities and explain core concepts from computational linguistics. The workshop was presented at numerous outlets in Italy between 2019 and 2020, both face-to-face and online","keywords":["nlp","teaching"],"pages":"160-170","url":"https:\/\/www.aclweb.org\/anthology\/2021.teachingnlp-1.26","volume":"","doi":"","editors":[],"editors_source":"","published":"Proceedings of the 5th Workshop on Teaching NLP","publisher":"","issn":"","isbn":"978-1-954085-36-7","conference_name":"5th Workshop on Teaching NLP","conference_place":"","conference_date":"","last_updated_cnr":"2024-12-22 03:11:32","last_updated_oai":"2024-12-22 03:11:32","last_updated_www":"0000-00-00 00:00:00"},{"id":2091,"id_source":400473,"institutes":["ILC","ISTI"],"type":"conference_article","type_order":7,"title":"How do BERT embeddings organize linguistic knowledge?","year":2021,"authors":["Puccetti, G.","Miaschi, A.","Dell'Orletta, F."],"authors_source":"Puccetti, G.; Miaschi, A.; Dell'Orletta, F.","authors_cnr_name":["PUCCETTI, GIOVANNI","MIASCHI, ALESSIO","DELL'ORLETTA, FELICE"],"authors_cnr_id":["rp13460","rp12522","rp22811"],"authors_cnr_institute":[],"abstract":"Several studies investigated the linguistic information implicitly encoded in Neural Language Models. Most of these works focused on quantifying the amount and type of information available within their internal representations and across their layers. In line with this scenario, we proposed a different study, based on Lasso regression, aimed at understanding how the information encoded by BERT sentence-level representations is arrange within its hidden units. Using a suite of several probing tasks, we showed the existence of a relationship between the implicit knowledge learned by the model and the number of individual units involved in the encodings of this competence. Moreover, we found that it is possible to identify groups of hidden units more relevant for specific linguistic properties","keywords":["NLP","Interpretability","Deep Learning"],"pages":"48-57","url":"https:\/\/www.aclweb.org\/anthology\/2021.deelio-1.6","volume":"","doi":"10.18653\/v1\/2021.deelio-1.6","editors":[],"editors_source":"","published":"Proceedings of the 2nd Workshop on Knowledge Extraction and Integrationfor Deep Learning Architectures","publisher":"","issn":"","isbn":"978-1-954085-30-5","conference_name":"2nd Workshop on Knowledge Extraction and Integrationfor Deep Learning Architectures","conference_place":"","conference_date":"","last_updated_cnr":"2025-02-05 23:04:33","last_updated_oai":"2025-02-05 23:04:33","last_updated_www":"0000-00-00 00:00:00"},{"id":1876,"id_source":421771,"institutes":["ILC"],"type":"conference_article","type_order":7,"title":"PRELEARN @ EVALITA 2020: Overview of the Prerequisite Relation Learning Task for Italian","year":2020,"authors":["Alzetta, C.","Miaschi, A.","Dell'Orletta, F.","Koceva","Frosina","Torre","Ilaria"],"authors_source":"Alzetta, Chiara; Alzetta, Chiara; Miaschi, Alessio; Miaschi, Alessio; Dell'Orletta, Felice; Dell'Orletta, Felice; Koceva, ; Frosina, ; Torre, ; Ilaria,","authors_cnr_name":["ALZETTA, CHIARA","ALZETTA, CHIARA","MIASCHI, ALESSIO","MIASCHI, ALESSIO","DELL'ORLETTA, FELICE","DELL'ORLETTA, FELICE"],"authors_cnr_id":["rp12530","rp12530","rp12522","rp12522","rp22811","rp22811"],"authors_cnr_institute":[],"abstract":"The Prerequisite Relation Learning (PRELEARN) task is the EVALITA 2020 shared task on concept prerequisite learning, which consists of classifying prerequisite relations between pairs of concepts distinguishing between prerequisite pairs and non-prerequisite pairs. Four sub-tasks were defined: two of them define different types of features that participants are allowed to use when training their model, while the other two define the classification scenarios where the proposed models would be tested. In total, 14 runs were submitted by 3 teams comprising 9 total individual participants","keywords":["nlp","prerequisite learning","shared task"],"pages":"","url":"http:\/\/ceur-ws.org\/Vol-2765\/paper164.pdf","volume":"","doi":"","editors":[],"editors_source":"","published":"Proceedings of the Seventh Evaluation Campaign of Natural Language Processing and Speech Tools for Italian (EVALITA)","publisher":"","issn":"","isbn":"","conference_name":"Seventh Evaluation Campaign of Natural Language Processing and Speech Tools for Italian (EVALITA)","conference_place":"","conference_date":"","last_updated_cnr":"2024-12-22 02:51:09","last_updated_oai":"2024-12-22 02:51:09","last_updated_www":"0000-00-00 00:00:00"},{"id":46,"id_source":421769,"institutes":["ILC"],"type":"conference_article","type_order":7,"title":"ATE ABSITA@ EVALITA2020: Overview of the Aspect Term Extraction and Aspect-based Sentiment Analysis Task","year":2020,"authors":["De Mattei, L.","De Martino, G.","Iovine, A.","Miaschi, A.","Polignano, M.","Rambelli, G."],"authors_source":"De, Mattei; Lorenzo, ; De, Martino; Graziella, ; Iovine, ; Andrea, ; Miaschi, Alessio; Miaschi, Alessio; Polignano, ; Marco, ; Rambelli, ; Giulia,","authors_cnr_name":["MIASCHI, ALESSIO","MIASCHI, ALESSIO"],"authors_cnr_id":["rp12522","rp12522"],"authors_cnr_institute":[],"abstract":"Over the last years, the rise of novel sentiment analysis techniques to assess aspect-based opinions on product reviews has become a key component for providing valuable insights to both consumers and businesses. To this extent, we propose ATE\\_ABSITA: the EVALITA 2020 shared task on Aspect Term Extraction and Aspect-Based Sentiment Analysis. In particular, we approach the task as a cascade of three subtasks: Aspect Term Extraction (ATE), Aspect-based Sentiment Analysis (ABSA) and Sentiment Analysis (SA). Therefore, we invited participants to submit systems designed to automatically identify the \"aspect terms\" in each review and to predict the sentiment expressed for each aspect, along with the sentiment of the entire review. The task received broad interest, with 27 teams registered and more than 45 participants. However, only three teams submitted their working systems. The results obtained underline the task's difficulty, but they also show how it is possible to deal with it using innovative approaches and models. Indeed, two of them are based on large pre-trained language models as typical in the current state of the art for the English language","keywords":["nlp","sentiment analysis","shared task"],"pages":"","url":"http:\/\/ceur-ws.org\/Vol-2765\/paper153.pdf","volume":"","doi":"","editors":[],"editors_source":"","published":"Proceedings of the Seventh Evaluation Campaign of Natural Language Processing and Speech Tools for Italian (EVALITA)","publisher":"","issn":"","isbn":"","conference_name":"Seventh Evaluation Campaign of Natural Language Processing and Speech Tools for Italian (EVALITA)","conference_place":"","conference_date":"","last_updated_cnr":"2024-12-21 06:05:34","last_updated_oai":"2024-12-21 06:05:34","last_updated_www":"0000-00-00 00:00:00"},{"id":1787,"id_source":421767,"institutes":["ILC"],"type":"conference_article","type_order":7,"title":"Is Neural Language Model Perplexity Related to Readability?","year":2020,"authors":["Miaschi, A.","Alzetta, C.","Brunato, D. P.","Dell'Orletta, F.","Venturi, G."],"authors_source":"Miaschi, Alessio; Miaschi, Alessio; Alzetta, Chiara; Alzetta, Chiara; Brunato, DOMINIQUE PIERINA; Brunato, DOMINIQUE PIERINA; Dell'Orletta, Felice; Dell'Orletta, Felice; Venturi, Giulia; Venturi, Giulia","authors_cnr_name":["MIASCHI, ALESSIO","MIASCHI, ALESSIO","ALZETTA, CHIARA","ALZETTA, CHIARA","BRUNATO, DOMINIQUE PIERINA","BRUNATO, DOMINIQUE PIERINA","DELL'ORLETTA, FELICE","DELL'ORLETTA, FELICE","VENTURI, GIULIA","VENTURI, GIULIA"],"authors_cnr_id":["rp12522","rp12522","rp12530","rp12530","rp06836","rp06836","rp22811","rp22811","rp00732","rp00732"],"authors_cnr_institute":[],"abstract":"This paper explores the relationship between Neural Language Model (NLM) perplexity and sentence readability. Starting from the evidence that NLMs implicitly acquire sophisticated linguistic knowledge from a huge amount of training data, our goal is to investigate whether perplexity is affected by linguistic features used to automatically assess sentence readability and if there is a correlation between the two metrics. Our findings suggest that this correlation is actually quite weak and the two metrics are affected by different linguistic phenomena","keywords":["nlp","neural language models","readability"],"pages":"","url":"http:\/\/ceur-ws.org\/Vol-2769\/paper_57.pdf","volume":"","doi":"","editors":[],"editors_source":"","published":"Proceedings of the Seventh Italian Conference on Computational Linguistics","publisher":"","issn":"","isbn":"979-12-80136-28-2","conference_name":"Seventh Italian Conference on Computational Linguistics","conference_place":"","conference_date":"","last_updated_cnr":"2024-12-20 02:13:07","last_updated_oai":"2024-12-20 02:13:07","last_updated_www":"0000-00-00 00:00:00"},{"id":1972,"id_source":379646,"institutes":["ILC"],"type":"conference_article","type_order":7,"title":"Linguistic Profiling of a Neural Language Model","year":2020,"authors":["Miaschi, A.","Brunato, D.","Dell'Orletta, F.","Venturi, G."],"authors_source":"Miaschi, A; Brunato, D; Dell'Orletta, F; Venturi, G","authors_cnr_name":["MIASCHI, ALESSIO","BRUNATO, DOMINIQUE PIERINA","DELL'ORLETTA, FELICE","VENTURI, GIULIA"],"authors_cnr_id":["rp12522","rp06836","rp22811","rp00732"],"authors_cnr_institute":[],"abstract":"In this paper we investigate the linguistic knowledge learned by a Neural Language Model (NLM) before and after a fine-tuning process and how this knowledge affects its predictions during several classification problems. We use a wide set of probing tasks, each of which corresponds to a distinct sentence-level feature extracted from different levels of linguistic annotation. We show that BERT is able to encode a wide range of linguistic characteristics, but it tends to lose this information when trained on specific downstream tasks. We also find that BERT's capacity to encode different kind of linguistic properties has a positive influence on its predictions: the more it stores readable linguistic information of a sentence, the higher will be its capacity of predicting the expected label assigned to that sentence","keywords":["Linguistic Profiling","Neural Language Model","Interpretability"],"pages":"745-756","url":"https:\/\/www.aclweb.org\/anthology\/2020.coling-main.65\/","volume":"","doi":"10.18653\/v1\/2020.coling-main.65","editors":[],"editors_source":"","published":"International Conference on Computational Linguistics (COLING)","publisher":"","issn":"","isbn":"978-1-952148-27-9","conference_name":"International Conference on Computational Linguistics (COLING)","conference_place":"","conference_date":"","last_updated_cnr":"2025-03-13 22:58:33","last_updated_oai":"2025-03-13 22:58:33","last_updated_www":"0000-00-00 00:00:00"},{"id":1056,"id_source":384933,"institutes":["ILC"],"type":"conference_article","type_order":7,"title":"Tracking the Evolution of Written Language Competence in L2 Spanish Learners","year":2020,"authors":["Miaschi, A.","Davidson, S.","Brunato, D. P.","Dell'Orletta, F.","Sagae, K.","Sanchez Gutierrez, C. H.","Venturi, G."],"authors_source":"Miaschi, Alessio; Davidson, Sam; Brunato, DOMINIQUE PIERINA; Dell'Orletta, Felice; Sagae, Kenji; SanchezGutierrez Claudia, H; Venturi, Giulia","authors_cnr_name":["MIASCHI, ALESSIO","BRUNATO, DOMINIQUE PIERINA","DELL'ORLETTA, FELICE","VENTURI, GIULIA"],"authors_cnr_id":["rp12522","rp06836","rp22811","rp00732"],"authors_cnr_institute":[],"abstract":"In this paper we present an NLP-based approach for tracking the evolution of written language competence in L2 Spanish learners using a wide range of linguistic features automatically extracted from students' written productions. Beyond reporting classification results for different scenarios, we explore the connection between the most predictive features and the teaching curriculum, finding that our set of linguistic features often reflects the explicit instruction that students receive during each course","keywords":["Evolution of Language Competence","Natural Language Processing","Linguistic Profiling"],"pages":"92-101","url":"https:\/\/www.aclweb.org\/anthology\/2020.bea-1.9.pdf","volume":"","doi":"10.18653\/v1\/W16-05","editors":[],"editors_source":"","published":"Proceedings of 15th Workshop on Innovative Use of NLP for Building Educational Applications","publisher":"Association for Computational Linguistics (Stroudsburg, USA)","issn":"","isbn":"978-1-941643-83-9","conference_name":"15th Workshop on Innovative Use of NLP for Building Educational Applications","conference_place":"Stroudsburg","conference_date":"","last_updated_cnr":"2025-03-19 22:34:12","last_updated_oai":"2025-03-19 22:34:12","last_updated_www":"0000-00-00 00:00:00"},{"id":1301,"id_source":421763,"institutes":["ILC"],"type":"conference_article","type_order":7,"title":"Contextual and Non-Contextual Word Embeddings: an in-depth Linguistic Investigation","year":2020,"authors":["Miaschi, A.","Dell'Orletta, F."],"authors_source":"Miaschi, Alessio; Miaschi, Alessio; Dell'Orletta, Felice; Dell'Orletta, Felice","authors_cnr_name":["MIASCHI, ALESSIO","MIASCHI, ALESSIO","DELL'ORLETTA, FELICE","DELL'ORLETTA, FELICE"],"authors_cnr_id":["rp12522","rp12522","rp22811","rp22811"],"authors_cnr_institute":[],"abstract":"In this paper we present a comparison between the linguistic knowledge encoded in the internal representations of a contextual Language Model (BERT) and a contextual-independent one (Word2vec). We use a wide set of probing tasks, each of which corresponds to a distinct sentence-level feature extracted from different levels of linguistic annotation. We show that, although BERT is capable of understanding the full context of each word in an input sequence, the implicit knowledge encoded in its aggregated sentence representations is still comparable to that of a contextual-independent model. We also find that BERT is able to encode sentence-level properties even within single-word embeddings, obtaining comparable or even superior results than those obtained with sentence representations","keywords":["nlp","interpretability","representation learning"],"pages":"110-119","url":"https:\/\/www.aclweb.org\/anthology\/2020.repl4nlp-1.15","volume":"","doi":"10.18653\/v1\/2020.repl4nlp-1.15","editors":[],"editors_source":"","published":"Proceedings of the 5th Workshop on Representation Learning for NLP","publisher":"","issn":"","isbn":"978-1-952148-15-6","conference_name":"5th Workshop on Representation Learning for NLP","conference_place":"","conference_date":"","last_updated_cnr":"2024-12-22 03:00:08","last_updated_oai":"2024-12-22 03:00:08","last_updated_www":"0000-00-00 00:00:00"},{"id":1129,"id_source":421765,"institutes":["ILC"],"type":"conference_article","type_order":7,"title":"Italian Transformers Under the Linguistic Lens","year":2020,"authors":["Miaschi, A.","Sarti, G.","Brunato, D. P.","Dell'Orletta, F.","Venturi, G."],"authors_source":"Miaschi, Alessio; Miaschi, Alessio; Sarti, ; Gabriele, ; Brunato, DOMINIQUE PIERINA; Brunato, DOMINIQUE PIERINA; Dell'Orletta, Felice; Dell'Orletta, Felice; Venturi, Giulia; Venturi, Giulia","authors_cnr_name":["MIASCHI, ALESSIO","MIASCHI, ALESSIO","BRUNATO, DOMINIQUE PIERINA","BRUNATO, DOMINIQUE PIERINA","DELL'ORLETTA, FELICE","DELL'ORLETTA, FELICE","VENTURI, GIULIA","VENTURI, GIULIA"],"authors_cnr_id":["rp12522","rp12522","rp06836","rp06836","rp22811","rp22811","rp00732","rp00732"],"authors_cnr_institute":[],"abstract":"In this paper we present an in-depth investigation of the linguistic knowledge encoded by the transformer models currently available for the Italian language. In particular, we investigate whether and how using different architectures of probing models affects the performance of Italian transformers in encoding a wide spectrum of linguistic features. Moreover, we explore how this implicit knowledge varies according to different textual genres","keywords":["nlp","neural language models","interpretability"],"pages":"","url":"http:\/\/ceur-ws.org\/Vol-2769\/paper_56.pdf","volume":"","doi":"","editors":[],"editors_source":"","published":"Proceedings of the Seventh Italian Conference on Computational Linguistics (CLiC-it)","publisher":"","issn":"","isbn":"979-12-80136-28-2","conference_name":"Seventh Italian Conference on Computational Linguistics (CLiC-it)","conference_place":"","conference_date":"","last_updated_cnr":"2024-12-20 02:17:38","last_updated_oai":"2024-12-20 02:17:38","last_updated_www":"0000-00-00 00:00:00"},{"id":2212,"id_source":390427,"institutes":["ILC"],"type":"conference_article","type_order":7,"title":"Prerequisite or not prerequisite? That's the problem! An NLP-based Approach for Concept Prerequisites Learning","year":2019,"authors":["Alzetta, C.","Miaschi, A.","Adorni, G.","Dell'Orletta, F.","Koceva, F.","Passalacqua, S.","Torre, I."],"authors_source":"Alzetta C.; Miaschi A.; Adorni G.; Dell'Orletta F.; Koceva F.; Passalacqua S.; Torre I.","authors_cnr_name":["ALZETTA, CHIARA","MIASCHI, ALESSIO","DELL'ORLETTA, FELICE"],"authors_cnr_id":["rp12530","rp12522","rp22811"],"authors_cnr_institute":[],"abstract":"This paper presents a method for prerequisite learning classification between educational concepts. The proposed system was developed by adapting a classification algorithm designed for sequencing Learning Objects to the task of ordering concepts from a computer science textbook. In order to apply the system to the new task, for each concept we automatically created a learning unit from the textbook using two criteria based on concept occurrences and burst intervals. Results are promising and suggest that further improvements could highly benefit the results","keywords":[],"pages":"","url":"https:\/\/iris.cnr.it\/handle\/20.500.14243\/390427","volume":"","doi":"","editors":[],"editors_source":"","published":"","publisher":"","issn":"","isbn":"9791280136008","conference_name":"","conference_place":"","conference_date":"","last_updated_cnr":"2025-06-14 00:55:21","last_updated_oai":"0000-00-00 00:00:00","last_updated_www":"0000-00-00 00:00:00"},{"id":1610,"id_source":390439,"institutes":["ILC"],"type":"conference_article","type_order":7,"title":"Linguistically-Driven Strategy for Concept Prerequisites Learning on Italian","year":2019,"authors":["Miaschi, A.","Alzetta, C.","Cardillo, F. A.","Dell'Orletta, F."],"authors_source":"Miaschi, Alessio; Alzetta, Chiara; Cardillo, FRANCO ALBERTO; Dell'Orletta, Felice","authors_cnr_name":["MIASCHI, ALESSIO","ALZETTA, CHIARA","CARDILLO, FRANCO ALBERTO","DELL'ORLETTA, FELICE"],"authors_cnr_id":["rp12522","rp12530","rp02590","rp22811"],"authors_cnr_institute":[],"abstract":"We present a new concept prerequisite learning method for Learning Object (LO) ordering that exploits only linguistic features extracted from textual educational resources. The method was tested in a cross-and in-domain scenario both for Italian and English. Additionally, we performed experiments based on a incremental training strategy to study the impact of the training set size on the classifier performances. The paper also introduces ITA-PREREQ, to the best of our knowledge the first Italian dataset annotated with prerequisite relations between pairs of educational concepts, and describe the automatic strategy devised to build it","keywords":["Concept Prerequisites Learning"],"pages":"285-295","url":"https:\/\/iris.cnr.it\/handle\/20.500.14243\/390439","volume":"","doi":"","editors":[],"editors_source":"","published":"Proceedings of the Fourteenth Workshop on Innovative Use of NLP for Building Educational Applications","publisher":"","issn":"","isbn":"","conference_name":"14th Workshop on Innovative Use of NLP for Building Educational Applications","conference_place":"","conference_date":"","last_updated_cnr":"2025-04-08 01:05:30","last_updated_oai":"2025-04-08 01:05:30","last_updated_www":"0000-00-00 00:00:00"},{"id":245,"id_source":493650,"institutes":["ILC","ISTI"],"type":"journal_article","type_order":1,"title":"The Codice Pelavicino between digital edition and Public History","year":2017,"authors":["Salvatori, E.","Rosselli Del Turco, R.","Alzetta, C.","Di Pietro, C.","Mannari, C.","Miaschi, A."],"authors_source":"Salvatori, E.; Rosselli Del Turco, R.; Alzetta, C.; Di Pietro, C.; Mannari, C.; Miaschi, A.","authors_cnr_name":["ALZETTA, CHIARA","MANNARI, CHIARA","MIASCHI, ALESSIO"],"authors_cnr_id":["rp12530","rp15892","rp12522"],"authors_cnr_institute":[],"abstract":"The Codice Pelavicino Digitale Project aims to publish an online digital edition of the relevant manuscript of the XIII century. In this paper features of the edition and related issues are addressed. Secondly we explain motivations for choosing a digital edition as a medium: we address the background, and common concerns in the context of Academy and clerical and historical archives. Finally we give insights on the international standard adopted to markup the text, i. e. XML-TEI, and EVT, a tool adopted to generate the final website and display texts and images","keywords":["Diplomatica","Filologia digitale","Latino medievale","Storia pubblica","TEI XML"],"pages":"105-117","url":"https:\/\/iris.cnr.it\/handle\/20.500.14243\/493650","volume":"2017 (1)","doi":"10.6092\/issn.2532-8816\/7232","editors":[],"editors_source":"","published":"UMANISTICA DIGITALE","publisher":"","issn":"2532-8816","isbn":"","conference_name":"","conference_place":"","conference_date":"","last_updated_cnr":"2024-09-30 23:17:24","last_updated_oai":"2024-09-30 23:17:24","last_updated_www":"0000-00-00 00:00:00"}]