[{"id":457666,"id_source":589603,"institutes":["ILC"],"type":"journal_article","type_order":1,"title":"What makes a book review compelling? Analyzing informativeness, writing style, and enjoyment","year":2026,"authors":["Alzetta, C.","Dell'Orletta, F.","Miaschi, A.","Venturi, G."],"authors_source":"Alzetta, Chiara; Dell'Orletta, Felice; Miaschi, Alessio; Venturi, Giulia","authors_cnr_name":["ALZETTA, CHIARA","DELL'ORLETTA, FELICE","MIASCHI, ALESSIO","VENTURI, GIULIA"],"authors_cnr_id":["rp12530","rp22811","rp12522","rp00732"],"authors_cnr_institute":[],"abstract":"Amateur book reviews published on Digital Social Reading (DSR) platforms play a crucial role in sharing reading experiences and capturing reader preferences. However, little attention has been given to the linguistic features that influence the way reading experiences are conveyed and shape readers\u2019 perceptions of the aspects discussed in reviews. This study addresses this gap by combining Computational Stylometry and Machine Learning techniques to examine how the linguistic features of book reviews combined with the demographic characteristics of review readers impact the reception of book reviews, focusing in particular on three key aspects that might make a book review compelling, namely informativeness, writing style, and enjoyment. To this aim, we relied on a corpus of Italian Goodreads reviews to investigate how amateur reviewers communicate their reading experience and on a survey to collect human judgments about review reception. Additionally, we investigated the extent to which the review style and the demographic characteristics of review readers can predict the different aspects of review perception. Our findings revealed that linguistic characteristics play a crucial role in shaping reader perceptions and are equally or even more predictive than demographic information in automatic classification models. Nevertheless, while demographic factors such as gender and birth year offer limited utility in forming homogeneous reader groups, reading habits emerged as a relevant factor in identifying shared trends among readers. These insights contribute to a deeper understanding of reader involvement in DSR communities and offer valuable insights for both publishers and review platforms in defining book recommender systems","keywords":["book reviews, human perception, reading experience, stylistic analysis, perception prediction"],"pages":"","url":"https:\/\/doi.org\/10.1093\/llc\/fqag080","volume":"","doi":"10.1093\/llc\/fqag080","editors":[],"editors_source":"","published":"DIGITAL SCHOLARSHIP IN THE HUMANITIES","publisher":"","issn":"2055-7671","isbn":"","conference_name":"","conference_place":"","conference_date":"","last_updated_cnr":"2026-07-09 00:17:49","last_updated_oai":"2026-07-09 00:17:49","last_updated_www":"0000-00-00 00:00:00"},{"id":457667,"id_source":589601,"institutes":["ILC"],"type":"journal_article","type_order":1,"title":"Teaming Up with Artificial Agents in Non-routine Analytical Tasks","year":2026,"authors":["Cominelli, L.","Andrea Galatolo, F.","Giannetti, C.","Dell'Orletta, F.","Ciaccio, C.","Chapkovski, P.","Venturi, G."],"authors_source":"Cominelli, Lorenzo; Andrea Galatolo, Federico; Giannetti, Caterina; Dell'Orletta, Felice; Ciaccio, Cristiano; Chapkovski, Philipp; Venturi, Giulia","authors_cnr_name":["DELL'ORLETTA, FELICE","Ciaccio, Cristiano","VENTURI, GIULIA"],"authors_cnr_id":["rp22811","rp27292","rp00732"],"authors_cnr_institute":[],"abstract":"Although AI systems are becoming increasingly common in the workplace, research on their integration into human teams remains limited. In particular, little is known about how the embodiment of artificial agents shapes collaboration and performance in non-routine analytical tasks. To address this gap, we examine how different degrees of embodiment affect team performance and conversational dynamics in a real-life escape room. Teams composed of either three humans or two humans and an artificial agent (a Box, an Avatar, or a hyper-realistic humanoid) worked together to escape the room within a time limit. Our findings show that artificial agents have an uneven impact on team outcomes, with some mixed human\u2013AI teams performing exceptionally well and others markedly worse. Human-only teams, by contrast, display more consistent performance: they are more likely to complete all tasks successfully, although they take longer and commit more errors. We also document a suggestive non-linear relationship between embodiment and team performance. Teams interacting with more embodied agents display conversational patterns that more closely resemble human\u2013human dialogue. Together, these findings show that embodied AI shapes collaboration in complex ways, reinforcing evidence that social cues critically guide teamwork dynamics","keywords":["human-AI collaboration, human-robot interaction, embodied AI, team performance, conversational dynamics, non-routine analytical tasks"],"pages":"","url":"https:\/\/doi.org\/10.1145\/3816428","volume":"","doi":"10.1145\/3816428","editors":[],"editors_source":"","published":"ACM TRANSACTIONS ON HUMAN-ROBOT INTERACTION","publisher":"","issn":"2573-9522","isbn":"","conference_name":"","conference_place":"","conference_date":"","last_updated_cnr":"2026-08-08 01:24:25","last_updated_oai":"2026-08-07 15:16:10","last_updated_www":"0000-00-00 00:00:00"},{"id":457057,"id_source":586463,"institutes":["ILC"],"type":"conference_article","type_order":7,"title":"Lexical Conditioning of Model's Distribution through Uncertainty-gated Soft-Mixing of Probabilities","year":2026,"authors":["Papucci, M.","Venturi, G.","Dell'Orletta, F."],"authors_source":"Papucci, Michele; Venturi, Giulia; Dell'Orletta, Felice","authors_cnr_name":["Papucci, Michele","VENTURI, GIULIA","DELL'ORLETTA, FELICE"],"authors_cnr_id":["rp28269","rp00732","rp22811"],"authors_cnr_institute":[],"abstract":"We present Uncertainty-Gated Lexical Decoding (UGLD), a decoding-time framework for fine-grained lexical control in Large Language Models (LLMs) that explicitly addresses the trade-off between controllability and fluency. UGLD adaptively scales intervention through an entropy-based gating mechanism derived from the model\u2019s predictive distribution, activating control when uncertainty is high and limiting interference when predictions are confident. The method supports both promotion toward and against predefined vocabularies. We evaluate UGLD in Italian on two open-weight LLMs (ANITA 8B and Qwen 3 4B) across paraphrasing and free-text generation settings, considering Simple Vocabulary Conditioning and Jargon Reduction scenarios. Automatic evaluation shows consistent improvements in lexical coverage over standard decoding strategies, while human evaluation confirms that fluency is preserved under controlled intervention","keywords":["Controlled Text Generation, Lexically Constrained Decoding, Entropy-Gated Decoding"],"pages":"89-100","url":"http:\/\/www.italianlp.it\/wp-content\/uploads\/2026\/05\/papuccietalreadixtsar2026.pdf","volume":"","doi":"","editors":[],"editors_source":"","published":"Proceedings of the Joint Workshop on Readability and Text Simplification (READIxTSAR) @ LREC 2026","publisher":"","issn":"","isbn":"978-2-493814-91-3","conference_name":"Joint Workshop on Readability and Text Simplification (READIxTSAR)","conference_place":"","conference_date":"","last_updated_cnr":"2026-07-09 00:17:43","last_updated_oai":"2026-07-08 13:37:11","last_updated_www":"0000-00-00 00:00:00"},{"id":1400,"id_source":580421,"institutes":["ILC"],"type":"conference_article","type_order":7,"title":"Controllable Sentence Simplification in Italian: Fine-Tuning Large Language Models on Automatically Generated Resources","year":2026,"authors":["Papucci, M.","Venturi, G.","Dell'Orletta, F."],"authors_source":"Papucci, Michele; Venturi, Giulia; Dell'Orletta, Felice","authors_cnr_name":["Papucci, Michele","VENTURI, GIULIA","DELL'ORLETTA, FELICE"],"authors_cnr_id":["rp28269","rp00732","rp22811"],"authors_cnr_institute":[],"abstract":"This paper presents a study on readability-controlled Sentence Simplification for Italian, addressing the scarcity of annotated resources for low-resource languages. We introduce IMPaCTS (Italian Multilevel Parallel Corpus for Text Simplification), the first fully automatically created corpus of 1, 444, 160 original\u2013simple sentence pairs automatically annotated with readability levels and linguistic features. It was generated using an Italian LLM prompted in zero-shot to produce multiple simplifications per input sentence. Increasing portions of the resource are used to fine-tune mono-and multilingual open-weight LLMs, conditioning them to generate simplifications at a target readability level. Results from automatic and human evaluations show that fine-tuning on IMPaCTS improves performance both in terms of task completion and adherence to the targeted readability levels compared to few-shot baselines","keywords":["Controlled Sentence Simplification, Readability Assessment, Large Language Models"],"pages":"7178-7191","url":"http:\/\/www.lrec-conf.org\/proceedings\/lrec2026\/pdf\/2026.lrec2026-1.570","volume":"","doi":"10.63317\/5fgm358dfxt5","editors":[],"editors_source":"","published":"Proceedings of the 15th Language Resources and Evaluation Conference (LREC 2026)","publisher":"","issn":"","isbn":"978-2-493814-49-4","conference_name":"15th Language Resources and Evaluation Conference (LREC 2026)","conference_place":"","conference_date":"","last_updated_cnr":"2026-07-09 00:17:44","last_updated_oai":"2026-07-09 00:17:44","last_updated_www":"0000-00-00 00:00:00"},{"id":1098,"id_source":570443,"institutes":["ILC"],"type":"journal_article","type_order":1,"title":"Parallel Trees: a novel resource with aligned dependency and constituency syntactic representations","year":2025,"authors":["Alzetta, C.","Miaschi, A.","Dell'Orletta, F.","Venturi, G.","Montemagni, S."],"authors_source":"Alzetta, C.; Miaschi, A.; Dell'Orletta, F.; Venturi, G.; Montemagni, S.","authors_cnr_name":["ALZETTA, CHIARA","MIASCHI, ALESSIO","DELL'ORLETTA, FELICE","VENTURI, GIULIA","MONTEMAGNI, SIMONETTA"],"authors_cnr_id":["rp12530","rp12522","rp22811","rp00732","rp16780"],"authors_cnr_institute":[],"abstract":"The paper introduces Parallel Trees, a novel multilingual treebank collection that includes 20 treebanks for 10 languages. The distinguishing property of this resource is that the sentences of each language are annotated using two syntactic representation paradigms (SRPs), respectively based on the notions of dependency and constituency. By aligning the annotations of existing resources, Parallel Trees represents an example of exploiting pre-existing treebanks to adapt them to novel applications. To illustrate its potential, we present a case study where the resource is employed as a benchmark to investigate whether and how BERT, one of the first prominent neural language models (NLMs), is sensitive to the dependency-and constituency-based approaches for representing the syntactic structure of a sentence. The case study results indicate that the model's sensitivity fluctuates across languages and experimental settings. The unique nature of the Parallel Trees resource creates the prerequisites for innovative studies comparing dependency and phrase-structure trees, allowing for more focused investigations without the interference of lexical variation","keywords":["Parallel treebanks","Syntactic representation","Diagnostic probing paradigm","Neural language model"],"pages":"3445-3485","url":"https:\/\/iris.cnr.it\/handle\/20.500.14243\/570443","volume":"59 (4)","doi":"10.1007\/s10579-025-09826-3","editors":[],"editors_source":"","published":"LANGUAGE RESOURCES AND EVALUATION","publisher":"","issn":"1574-020X","isbn":"","conference_name":"","conference_place":"","conference_date":"","last_updated_cnr":"2026-03-04 01:27:44","last_updated_oai":"2026-03-04 01:27:44","last_updated_www":"0000-00-00 00:00:00"},{"id":25,"id_source":484441,"institutes":["ILC"],"type":"journal_article","type_order":1,"title":"Parlamint-it: an 18-karat UD treebank of Italian parliamentary speeches","year":2025,"authors":["Alzetta, C.","Montemagni, S.","Sartor, M.","Venturi, G."],"authors_source":"Alzetta, Chiara; Montemagni, Simonetta; Sartor, Marta; Venturi, Giulia","authors_cnr_name":["ALZETTA, CHIARA","MONTEMAGNI, SIMONETTA","VENTURI, GIULIA"],"authors_cnr_id":["rp12530","rp16780","rp00732"],"authors_cnr_institute":[],"abstract":"The paper presents ParlaMint-It, a new treebank of Italian parliamentary debates, linguistically annotated based on the Universal Dependencies (UD) framework. The resource comprises 20, 460 tokens and represents a hybrid language variety that is underrepresented in the UD initiative. ParlaMint-It results from a manual revision process that relies on a semi-automatic methodology able to identify sentences that are most likely to contain inconsistencies and recurrent error patterns generated by the automatic annotation. Such a method made the revision process faster and more efficient than revising the entire treebank. In addition, it allowed the identification and correction of annotation errors resulting from linguistic constructions inconsis-tently represented in UD treebanks and from characteristics specific to parliamentary speeches. Hence, the treebank is deemed as an 18-karat resource, since, although not fully manually revised, it is a valuable resource for researchers working on Italian language processing tasks","keywords":["Universal dependencies treebanks, Annotation revision, Italian parliamentary debates, Linguistic annotation"],"pages":"1659-1683","url":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s10579-024-09748-6.pdf","volume":"59","doi":"10.1007\/s10579-024-09748-6","editors":[],"editors_source":"","published":"LANGUAGE RESOURCES AND EVALUATION","publisher":"","issn":"1574-020X","isbn":"","conference_name":"","conference_place":"","conference_date":"","last_updated_cnr":"2026-07-09 00:17:43","last_updated_oai":"2026-07-09 00:17:43","last_updated_www":"0000-00-00 00:00:00"},{"id":1997,"id_source":560370,"institutes":["ILC","IGSG"],"type":"journal_article","type_order":1,"title":"Metodi e strumenti per la modernizzazione della lingua delle istituzioni","year":2025,"authors":["Romano, F.","Venturi, G."],"authors_source":"Romano, Francesco; Venturi, Giulia","authors_cnr_name":["ROMANO, FRANCESCO","VENTURI, GIULIA"],"authors_cnr_id":["rp14613","rp00732"],"authors_cnr_institute":[],"abstract":"Abstract The paper provides a methodology for the guided drafting of administrative documents that meet criteria of clarity and simplicity in both content and language","keywords":["plain language, legal design, natural language processing","inclusione sociale, linguaggio giuridico, semplificazione linguaggio giuridico"],"pages":"105-121","url":"https:\/\/calumet-review.com\/index.php\/it\/category\/numeri\/22-s-sem-2025-it\/","volume":"MIGRANTI LEGGI CONTRATTI VERSO LA CHIAREZZA (EDITOR ANNARITA MIGLIETTA) (22)","doi":"","editors":[],"editors_source":"","published":"CALUMET","publisher":"","issn":"2465-0145","isbn":"","conference_name":"","conference_place":"","conference_date":"","last_updated_cnr":"2026-01-09 01:29:57","last_updated_oai":"2026-01-09 01:29:57","last_updated_www":"0000-00-00 00:00:00"},{"id":273,"id_source":570782,"institutes":["ILC"],"type":"journal_article","type_order":1,"title":"From clicks to care: Exploring the digital strategies of Italian health authorities in communicating \u2018General Practitioner Selection\u2019 service","year":2025,"authors":["Vinci, A.","Pirrotta, L.","Venturi, G.","Vainieri, M."],"authors_source":"Alessandro Vinci, Luca Pirrotta, Giulia Venturi, Milena Vainieri","authors_cnr_name":[],"authors_cnr_id":[],"authors_cnr_institute":[],"abstract":"The prioritization of digitalization is crucial to the agendas of nations worldwide. While substantial funds have been allocated to foster it, there remains a scarcity of tools dedicated to systematically monitoring the performance of the digital transformation. This work describes the level of digitalization and information of a fundamental primary care service: the \u201cGeneral Practitioner (GP) selection\u201d. The analysis was conducted by consulting websites of Italian Local Health Authorities (LHAs). First, we explored the digitalization levels of 105 websites through the Primary Care Digital Information (PCDI) composite index. It comprises four dimensions: informativeness, accessibility, inclusiveness, and adaptability, scoring on a five-point scale (low-high digitalization). Second, we conducted a readability analysis, employing three validated measures. We found an average level of digitalization and information, although dimensions perform differently. The best-performing dimension was adaptability, while the worst was inclusiveness. Half of the LHAs provided several digital alternatives to GP selection, while the remaining provided limited or no options. Regarding readability, just 29% of the LHA's websites were found easy to read. Overall, our findings depict that Italian LHAs have different approaches. This study highlights that, despite best practices, several areas require monitoring and intervention. Moreover, some barriers characterize Italian health communication strategies, notably the variability of information across and within regions and on average low website readability","keywords":["Online information, Readability, Evaluation, Digital transformation, Public services"],"pages":"9","url":"https:\/\/www.sciencedirect.com\/science\/article\/pii\/S0168851025001034","volume":"157","doi":"10.1016\/j","editors":[],"editors_source":"","published":"HEALTH POLICY","publisher":"","issn":"","isbn":"","conference_name":"","conference_place":"","conference_date":"","last_updated_cnr":"0000-00-00 00:00:00","last_updated_oai":"2026-03-04 01:27:53","last_updated_www":"0000-00-00 00:00:00"},{"id":857,"id_source":570801,"institutes":["ILC"],"type":"conference_article","type_order":7,"title":"Generating and Evaluating Multi-Level Text Simplification: A Case Study on Italian","year":2025,"authors":["Papucci, M.","Venturi, G.","Dell'Orletta, F."],"authors_source":"Michele Papucci, Giulia Venturi, Felice Dell'Orletta","authors_cnr_name":[],"authors_cnr_id":[],"authors_cnr_institute":[],"abstract":"Recent advances in Generative AI and Large Language Models (LLMs) have enabled the creation of highly realistic synthetic content, yet controlling model outputs remains a challenge. In this study, we explore the use of LLMs to generate high-quality synthetic data for Automatic Text Simplification (ATS), evaluating the ability of models fine-tuned on Italian to produce multiple simplified versions of the same original sentence that vary in readability and in their lexical and (morpho-)syntactic characteristics. The approach is tested across two domains, Wikipedia and Public Administration, allowing us to explore domain sensitivity. Additionally, we compare the linguistic phenomena observed in the generated data with those found in ATS resources previously created through manual or semi-automatic methods. Our results suggest that the best-performing LLM can generate linguistically diverse simplifications that align with known simplification patterns, offering a promising direction for building reliable ATS resources, including simplifications suited to varying levels of reader proficiency","keywords":["Automatic Text Simplification, Large Language Models, Synthetic Data, Linguistic Complexity, Sentence Readability"],"pages":"870-885","url":"https:\/\/aclanthology.org\/2025.clicit-1.82\/","volume":"","doi":"","editors":[],"editors_source":"","published":"Proceedings of the Eleventh Italian Conference on Computational Linguistics (CLiC-it 2025)","publisher":"CEUR Workshop Proceedings","issn":"","isbn":"979-12-243-0587-3","conference_name":"Eleventh Italian Conference on Computational Linguistics (CLiC-it 2025)","conference_place":"","conference_date":"","last_updated_cnr":"0000-00-00 00:00:00","last_updated_oai":"2026-03-03 18:40:13","last_updated_www":"0000-00-00 00:00:00"},{"id":437,"id_source":487005,"institutes":["ILC"],"type":"conference_article","type_order":7,"title":"Linguistic Knowledge Can Enhance Encoder-Decoder Models (If You Let It)","year":2024,"authors":["Miaschi, A.","Dell'Orletta, F.","Venturi, G."],"authors_source":"Miaschi, Alessio; Dell'Orletta, Felice; Venturi, Giulia","authors_cnr_name":["MIASCHI, ALESSIO","DELL'ORLETTA, FELICE","VENTURI, GIULIA"],"authors_cnr_id":["rp12522","rp22811","rp00732"],"authors_cnr_institute":[],"abstract":"In this paper, we explore the impact of augmenting pre-trained Encoder-Decoder models, specifically T5, with linguistic knowledge for the prediction of a target task. In particular, we investigate whether fine-tuning a T5 model on an intermediate task that predicts structural linguistic properties of sentences modifies its performance in the target task of predicting sentence-level complexity. Our study encompasses diverse experiments conducted on Italian and English datasets, employing both monolingual and multilingual T5 models at various sizes. Results obtained for both languages and in cross-lingual configurations show that linguistically motivated intermediate fine-tuning has generally a positive impact on target task performance, especially when applied to smaller models and in scenarios with limited data availability","keywords":["encoder-decoder, intermediate fine-tuning, linguistic features, sentence complexity"],"pages":"10539-10554","url":"https:\/\/aclanthology.org\/2024.lrec-main.922\/","volume":"","doi":"","editors":[],"editors_source":"","published":"Proceedings of the 2024 Joint International Conference on Computational Linguistics, Language Resources and Evaluation (LREC-COLING 2024)","publisher":"ELRA and ICCL","issn":"","isbn":"978-2-493814-10-4","conference_name":"Joint International Conference on Computational Linguistics, Language Resources and Evaluation (LREC-COLING 2024)","conference_place":"","conference_date":"","last_updated_cnr":"2025-03-05 05:11:23","last_updated_oai":"2025-03-05 05:11:23","last_updated_www":"0000-00-00 00:00:00"},{"id":540,"id_source":518427,"institutes":["ILC"],"type":"conference_article","type_order":7,"title":"Evaluating Large Language Models via Linguistic Profiling","year":2024,"authors":["Miaschi, A.","Dell'Orletta, F.","Venturi, G."],"authors_source":"Miaschi, Alessio; Dell'Orletta, Felice; Venturi, Giulia","authors_cnr_name":["MIASCHI, ALESSIO","DELL'ORLETTA, FELICE","VENTURI, GIULIA"],"authors_cnr_id":["rp12522","rp22811","rp00732"],"authors_cnr_institute":[],"abstract":"Large Language Models (LLMs) undergo extensive evaluation against various benchmarks collected in established leaderboards to assess their performance across multiple tasks. However, to the best of our knowledge, there is a lack of comprehensive studies evaluating these models\u2019 linguistic abilities independent of specific tasks. In this paper, we introduce a novel evaluation methodology designed to test LLMs\u2019 sentence generation abilities under specific linguistic constraints. Drawing on the \u2018linguistic profiling\u2019 approach, we rigorously investigate the extent to which five LLMs of varying sizes, tested in both zero-and few-shot scenarios, effectively adhere to (morpho)syntactic constraints. Our findings shed light on the linguistic proficiency of LLMs, revealing both their capabilities and limitations in generating linguistically-constrained sentences","keywords":["Large Language Models, Controllable Text Generation, Linguistic Profiling"],"pages":"2835-2848","url":"https:\/\/aclanthology.org\/2024.emnlp-main.166","volume":"","doi":"10.18653\/v1\/2024.emnlp-main.166","editors":[],"editors_source":"","published":"Proceedings of the 2024 Conference on Empirical Methods in Natural Language Processing","publisher":"Association for Computational Linguistics (USA)","issn":"","isbn":"979-8-89176-164-3","conference_name":"Conference on Empirical Methods in Natural Language Processing (EMNLP)","conference_place":"USA","conference_date":"","last_updated_cnr":"2025-02-08 06:50:29","last_updated_oai":"2025-02-08 06:50:29","last_updated_www":"0000-00-00 00:00:00"},{"id":726,"id_source":483001,"institutes":["ILC","IGSG"],"type":"misc","type_order":11,"title":"Linguistically annotated multilingual comparable corpora of parliamentary debates ParlaMint. ana 4. 1","year":2024,"authors":["Erjavec, T.","Kopp, M.","Ogrodniczuk, M.","Osenova, P.","Agerri, R.","Agirrezabal, M.","Agnoloni, T.","Aires, J.","Albini, M.","Alkorta, J.","Antiba Cartazo, I.","Arrieta, E.","Barcala, M.","Bardanca, D.","Barkarson, S.","Bartolini, R.","Battistoni, R.","Bel, N.","Bonet Ramos, M. D. M.","Calzada P\u00e9rez, M.","Cardoso, A.","\u00c7\u00f6ltekin, \u00c7.","Coole, M.","Dar\u0123is, R.","De Does, J.","De Libano, R.","Depoorter, G.","Depuydt, K.","Diwersy, S.","Dod\u00e9, R.","Fernandez, K.","Fern\u00e1ndez Rei, E.","Frontini, F.","Garcia, M.","Garc\u00eda D\u00edaz, N.","Garc\u00eda Louzao, P.","Gavriilidou, M.","Gkoumas, D.","Grigorov, I.","Grigorova, V.","Haltrup Hansen, D.","Iruskieta, M.","Jarlbrink, J.","Jelencsik M\u00e1tyus, K.","Jongejan, B.","Kahusk, N.","Kirnbauer, M.","Kryvenko, A.","Ligeti Nagy, N.","Ljube\u0161i\u0107, N.","Luxardo, G.","Magari\u00f1os, C.","Magnusson, M.","Marchetti, C.","Marx, M.","Meden, K.","Mendes, A.","Mochtak, M.","M\u00f6lder, M.","Montemagni, S.","Navarretta, C.","Nito\u0144, B.","Nor\u00e9n, F. M.","Nwadukwe, A.","Ojster\u0161ek, M.","Pan\u010dur, A.","Papavassiliou, V.","Pereira, R.","P\u00e9rez Lago, M.","Piperidis, S.","Pirker, H.","Pisani, M.","Pol, H. V. D.","Prokopidis, P.","Quochi, V.","Rayson, P.","Regueira, X. L.","Rii, A.","Rudolf, M.","Ruisi, M.","Rupnik, P.","Schopper, D.","Simov, K.","Sinikallio, L.","Skubic, J.","Tamper, M.","Tungland, L. M.","Tuominen, J.","Van Heusden, R.","Varga, Z.","V\u00e1zquez Abu\u00edn, M.","Venturi, G.","Vidal Migu\u00e9ns, A.","Vider, K.","Vivel Couso, A.","Vladu, A. I.","Wissik, T.","Yrj\u00e4n\u00e4inen, V.","Zevallos, R.","Fi\u0161er, D."],"authors_source":"Erjavec, Toma\u017e; Kopp, Maty\u00e1\u0161; Ogrodniczuk, Maciej; Osenova, Petya; Agerri, Rodrigo; Agirrezabal, Manex; Agnoloni, Tommaso; Aires, Jos\u00e9; Albini, Monica; Alkorta, Jon; Antiba-Cartazo, Iv\u00e1n; Arrieta, Ekain; Barcala, Mario; Bardanca, Daniel; Barkarson, Starka\u00f0ur; Bartolini, Roberto; Battistoni, Roberto; Bel, Nuria; Bonet Ramos, Maria del Mar; Calzada P\u00e9rez, Mar\u00eda; Cardoso, Aida; \u00c7\u00f6ltekin, \u00c7a\u011fr\u0131; Coole, Matthew; Dar\u0123is, Roberts; de Does, Jesse; de Libano, Ruben; Depoorter, Griet; Depuydt, Katrien; Diwersy, Sascha; Dod\u00e9, R\u00e9ka; Fernandez, Kike; Fern\u00e1ndez Rei, Elisa; Frontini, Francesca; Garcia, Marcos; Garc\u00eda D\u00edaz, Noelia; Garc\u00eda Louzao, Pedro; Gavriilidou, Maria; Gkoumas, Dimitris; Grigorov, Ilko; Grigorova, Vladislava; Haltrup Hansen, Dorte; Iruskieta, Mikel; Jarlbrink, Johan; Jelencsik-M\u00e1tyus, Kinga; Jongejan, Bart; Kahusk, Neeme; Kirnbauer, Martin; Kryvenko, Anna; Ligeti-Nagy, No\u00e9mi; Ljube\u0161i\u0107, Nikola; Luxardo, Giancarlo; Magari\u00f1os, Carmen; Magnusson, M\u00e5ns; Marchetti, Carlo; Marx, Maarten; Meden, Katja; Mendes, Am\u00e1lia; Mochtak, Michal; M\u00f6lder, Martin; Montemagni, Simonetta; Navarretta, Costanza; Nito\u0144, Bart\u0142omiej; Nor\u00e9n, Fredrik Mohammadi; Nwadukwe, Amanda; Ojster\u0161ek, Mihael; Pan\u010dur, Andrej; Papavassiliou, Vassilis; Pereira, Rui; P\u00e9rez Lago, Mar\u00eda; Piperidis, Stelios; Pirker, Hannes; Pisani, Marilina; Pol, Henk van der; Prokopidis, Prokopis; Quochi, Valeria; Rayson, Paul; Regueira, Xos\u00e9 Lu\u00eds; Rii, Andriana; Rudolf, Micha\u0142; Ruisi, Manuela; Rupnik, Peter; Schopper, Daniel; Simov, Kiril; Sinikallio, Laura; Skubic, Jure; Tamper, Minna; Tungland, Lars Magne; Tuominen, Jouni; van Heusden, Ruben; Varga, Zs\u00f3fia; V\u00e1zquez Abu\u00edn, Marta; Venturi, Giulia; Vidal Migu\u00e9ns, Adri\u00e1n; Vider, Kadri; Vivel Couso, Ainhoa; Vladu, Adina Ioana; Wissik, Tanja; Yrj\u00e4n\u00e4inen, V\u00e4in\u00f6; Zevallos, Rodolfo; Fi\u0161er, Darja","authors_cnr_name":["AGNOLONI, TOMMASO","BARTOLINI, ROBERTO","FRONTINI, FRANCESCA","MONTEMAGNI, SIMONETTA","QUOCHI, VALERIA","VENTURI, GIULIA"],"authors_cnr_id":["rp21506","rp00239","rp02790","rp16780","rp13283","rp00732"],"authors_cnr_institute":[],"abstract":"ParlaMint 4. 1 is a set of comparable corpora containing transcriptions of parliamentary debates of 29 European countries and autonomous regions, mostly starting in 2015 and extending to mid-2022. The individual corpora comprise between 9 and 126 million words and the complete set contains over 1. 2 billion words. The transcriptions are divided by days with information on the term, session and meeting, and contain speeches marked by the speaker and their role (e. g. chair, regular speaker). The speeches also contain marked-up transcriber comments, such as gaps in the transcription, interruptions, applause, etc. The corpora have extensive metadata, most importantly on speakers (name, gender, MP and minister status, party affiliation), on their political parties and parliamentary groups (name, coalition\/opposition status, Wikipedia-sourced left-to-right political orientation, and CHES variables, https: \/\/www. chesdata. eu\/). Note that some corpora have further metadata, e. g. the year of birth of the speakers, links to their Wikipedia articles, their membership in various committees, etc. The transcriptions are also marked with the subcorpora they belong to (\"reference\", until 2020-01-30, \"covid\", from 2020-01-31, and \"war\", from 2022-02-24). An overview of the statistics of the corpora is avaialable on GitHub in the folder Build\/Metadata, in particular for the release 4. 1 at https: \/\/github. com\/clarin-eric\/ParlaMint\/tree\/v4. 1\/Build\/Metadata. The corpora are encoded according to the ParlaMint encoding guidelines (https: \/\/clarin-eric. github. io\/ParlaMint\/) and schemas (included in the distribution). The ParlaMint. ana linguistic annotation includes tokenization; sentence segmentation; lemmatisation; Universal Dependencies part-of-speech, morphological features, and syntactic dependencies; and the 4-class CoNLL-2003 named entities. Some corpora also have further linguistic annotations, in particular PoS tagging according a language-specific scheme, with their corpus TEI headers giving further details on the annotation vocabularies and tools used. This entry contains the ParlaMint. ana TEI-encoded linguistically annotated corpora; the derived CoNLL-U files along with TSV metadata of the speeches; and the derived vertical files (with their registry file), suitable for use with CQP-based concordancers, such as CWB, noSketch Engine or KonText. Also included is the 4. 1 release of the sample data and scripts available at the GitHub repository of the ParlaMint project at https: \/\/github. com\/clarin-eric\/ParlaMint and the log files produced in the process of building the corpora for this release. The log files show e. g. known errors in the corpora, while more information about known problems is available in the open issues at the GitHub repository of the project. This entry contains the linguistically marked-up version of the corpus, while the text version, i. e. without the linguistic annotation is also available at http: \/\/hdl. handle. net\/11356\/1912. Another related resource, namely the ParlaMint corpora machine translated to English ParlaMint-en. ana 4. 1 can be found at http: \/\/hdl. handle. net\/11356\/1910. As opposed to the previous version 4. 0, this version fixes a number of bugs and restructures the ParlaMint GitHub repository. The DK corpus has been linguistically re-annotated to remove bugs, while its speeches are now also marked with topics. The PT corpus has been extended to 2024-03 and the UA corpus to 2023-11, which also has improved language marking (uk vs. ru) on segments","keywords":["ParlaCLARIN, linguistic annotation, pos-tagging, Named Entity Recognition, linguistic dependency annotation, UD"],"pages":"","url":"https:\/\/iris.cnr.it\/handle\/20.500.14243\/483001","volume":"","doi":"","editors":[],"editors_source":"","published":"","publisher":"","issn":"","isbn":"","conference_name":"","conference_place":"","conference_date":"","last_updated_cnr":"2025-03-07 06:08:51","last_updated_oai":"2025-03-07 06:08:51","last_updated_www":"0000-00-00 00:00:00"},{"id":1435,"id_source":439017,"institutes":["ILC"],"type":"journal_article","type_order":1,"title":"Tell me how you write and I'll tell you what you read: a study on the writing style of book reviews","year":2023,"authors":["Alzetta, C.","Dell'Orletta, F.","Miaschi, A.","Prat, E.","Venturi, G."],"authors_source":"Alzetta, Chiara; Dell'Orletta, Felice; Miaschi, Alessio; Prat, Elena; Venturi, Giulia","authors_cnr_name":["ALZETTA, CHIARA","DELL'ORLETTA, FELICE","MIASCHI, ALESSIO","VENTURI, GIULIA"],"authors_cnr_id":["rp12530","rp22811","rp12522","rp00732"],"authors_cnr_institute":[],"abstract":"The paper aims at investigating variations in the writing style of book reviews published on different social reading platforms and referring to books of different genres, which enables acquiring insights into communication strategies adopted by readers to share their reading experiences. To this end, we introduce a corpus-based study focused on the analysis of A Good Review, a novel corpus of online book reviews written in Italian, posted on Amazon and Goodreads, and covering six literary fiction genres. We rely on stylometric analysis to explore the linguistic properties and lexicon of reviews and the authors conducted automatic classification experiments using multiple approaches and feature configurations to predict either the review's platform or the literary genre. The analysis of user-generated reviews demonstrates that language is a quite variable dimension across reading platforms, but not as much across book genres. The classification experiments revealed that features modelling the syntactic structure of the sentence are reliable proxies for discerning Amazon and Goodreads reviews, whereas lexical information showed a higher predictive role for automatically discriminating the genre","keywords":["Stylometric analysis","Textual Genre detection","Book reviews"],"pages":"23","url":"https:\/\/www.emerald.com\/insight\/content\/doi\/10.1108\/JD-04-2023-0073\/full\/html","volume":"79","doi":"10.1108\/JD-04-2023-0073","editors":[],"editors_source":"","published":"JOURNAL OF DOCUMENTATION","publisher":"","issn":"0022-0418","isbn":"","conference_name":"","conference_place":"","conference_date":"","last_updated_cnr":"2025-03-06 04:29:16","last_updated_oai":"2025-03-06 04:29:16","last_updated_www":"0000-00-00 00:00:00"},{"id":1994,"id_source":448001,"institutes":["ILC","IGSG"],"type":"journal_article","type_order":1,"title":"The ParlaMint corpora of parliamentary proceedings","year":2023,"authors":["Erjavec, T.","Ogrodniczuk, M.","Osenova, P.","Ljubesic, N.","Simov, K.","Pancur, A.","Rudolf, M.","Kopp, M.","Barkarson, S.","Steingrimsson, S.","Coltekin, C.","De Does, J.","Depuydt, K.","Agnoloni, T.","Venturi, G.","Perez, M. C.","De Macedo, L. D.","Navarretta, C.","Luxardo, G.","Coole, M.","Rayson, P.","Morkevicius, V.","Krilavicius, T.","Dargis, R.","Ring, O.","Van Heusden, R.","Marx, M.","Fiser, D."],"authors_source":"Erjavec T.; Ogrodniczuk M.; Osenova P.; Ljubesic N.; Simov K.; Pancur A.; Rudolf M.; Kopp M.; Barkarson S.; Steingrimsson S.; Coltekin C.; de Does J.; Depuydt K.; Agnoloni T.; Venturi G.; Perez M.C.; de Macedo L.D.; Navarretta C.; Luxardo G.; Coole M.; Rayson P.; Morkevicius V.; Krilavicius T.; Dargis R.; Ring O.; van Heusden R.; Marx M.; Fiser D.","authors_cnr_name":["AGNOLONI, TOMMASO","VENTURI, GIULIA"],"authors_cnr_id":["rp21506","rp00732"],"authors_cnr_institute":[],"abstract":"This paper presents the ParlaMint corpora containing transcriptions of the sessions of the 17 European national parliaments with half a billion words. The corpora are uniformly encoded, contain rich meta-data about 11 thousand speakers, and are linguistically annotated following the Universal Dependencies formalism and with named entities. Samples of the corpora and conversion scripts are available from the project's GitHub repository, and the complete corpora are openly available via the CLARIN. SI repository for download, as well as through the NoSketch Engine and KonText concordancers and the Parlameter interface for on-line exploration and analysis","keywords":["Parlamentary proceedings","Linguistic annotation","Universal Dependencies"],"pages":"1-34","url":"http:\/\/www.scopus.com\/record\/display.url?eid=2-s2.0-85124105199&origin=inward","volume":"","doi":"10.1007\/s10579-021-09574-0","editors":[],"editors_source":"","published":"LANGUAGE RESOURCES AND EVALUATION","publisher":"","issn":"1574-020X","isbn":"","conference_name":"","conference_place":"","conference_date":"","last_updated_cnr":"2024-07-23 11:21:26","last_updated_oai":"2024-07-23 11:21:26","last_updated_www":"0000-00-00 00:00:00"},{"id":2064,"id_source":439018,"institutes":["ILC"],"type":"journal_article","type_order":1,"title":"Testing the Effectiveness of the Diagnostic Probing Paradigm on Italian Treebanks","year":2023,"authors":["Miaschi, A.","Alzetta, C.","Brunato, D.","Dell'Orletta, F.","Venturi, G."],"authors_source":"Miaschi, Alessio; Alzetta, Chiara; Brunato, Dominique; Dell'Orletta, Felice; Venturi, Giulia","authors_cnr_name":["MIASCHI, ALESSIO","ALZETTA, CHIARA","BRUNATO, DOMINIQUE PIERINA","DELL'ORLETTA, FELICE","VENTURI, GIULIA"],"authors_cnr_id":["rp12522","rp12530","rp06836","rp22811","rp00732"],"authors_cnr_institute":[],"abstract":"The outstanding performance recently reached by neural language models (NLMs) across many natural language processing (NLP) tasks has steered the debate towards understanding whether NLMs implicitly learn linguistic competence. Probes, i. e., supervised models trained using NLM representations to predict linguistic properties, are frequently adopted to investigate this issue. However, it is still questioned if probing classification tasks really enable such investigation or if they simply hint at surface patterns in the data. This work contributes to this debate by presenting an approach to assessing the effectiveness of a suite of probing tasks aimed at testing the linguistic knowledge implicitly encoded by one of the most prominent NLMs, BERT. To this aim, we compared the performance of probes when predicting gold and automatically altered values of a set of linguistic features. Our experiments were performed on Italian and were evaluated across BERT's layers and for sentences with different lengths. As a general result, we observed higher performance in the prediction of gold values, thus suggesting that the probing model is sensitive to the distortion of feature values. However, our experiments also showed that the length of a sentence is a highly influential factor that is able to confound the probing model's predictions","keywords":["Neural language model","Probing tasks","Treebanks"],"pages":"19","url":"https:\/\/www.mdpi.com\/2078-2489\/14\/3\/144","volume":"14 (3)","doi":"10.3390\/info14030144","editors":[],"editors_source":"","published":"INFORMATION","publisher":"","issn":"2078-2489","isbn":"","conference_name":"","conference_place":"","conference_date":"","last_updated_cnr":"2025-03-14 22:18:11","last_updated_oai":"2025-03-14 22:18:11","last_updated_www":"0000-00-00 00:00:00"},{"id":1544,"id_source":470901,"institutes":["ILC"],"type":"conference_article","type_order":7,"title":"LangLearn at EVALITA 2023: Overview of the Language Learning Development Task","year":2023,"authors":["Alzetta, C.","Brunato, D.","Dell'Orletta, F.","Miaschi, A.","Sagae, K.","S\u00e1nchez Guti\u00e9rrez, C. H.","Venturi, G."],"authors_source":"Alzetta, Chiara; Brunato, Dominique; Dell'Orletta, Felice; Miaschi, Alessio; Sagae, Kenji; S\u00e1nchez-Guti\u00e9rrez, Claudia H.; Venturi, Giulia","authors_cnr_name":["ALZETTA, CHIARA","BRUNATO, DOMINIQUE PIERINA","DELL'ORLETTA, FELICE","MIASCHI, ALESSIO","VENTURI, GIULIA"],"authors_cnr_id":["rp12530","rp06836","rp22811","rp12522","rp00732"],"authors_cnr_institute":[],"abstract":"Language Learning Development (LangLearn) is the EVALITA 2023 shared task on automatic language development assessment, which consists in predicting the evolution of the written language abilities of learners across time. LangLearn is conceived to be multilingual, relying on written productions of Italian and Spanish learners, and representative of L1 and L2 learning scenarios. A total of 9 systems were submitted by 5 teams. The results highlight the open challenges of automatic language development assessment","keywords":[],"pages":"","url":"https:\/\/iris.cnr.it\/handle\/20.500.14243\/470901","volume":"","doi":"","editors":[],"editors_source":"","published":"Proceedings of EVALITA 2023","publisher":"Accademia University Press (Torino, ITA)","issn":"","isbn":"9791255000693","conference_name":"8th Evaluation Campaign of Natural Language Processing and Speech Tools for Italian","conference_place":"Torino","conference_date":"","last_updated_cnr":"2025-01-24 23:18:36","last_updated_oai":"2025-01-24 23:18:36","last_updated_www":"0000-00-00 00:00:00"},{"id":1525,"id_source":470921,"institutes":["ILC"],"type":"conference_article","type_order":7,"title":"Unmasking the Wordsmith: Revealing Author Identity through Reader Reviews","year":2023,"authors":["Alzetta, C.","Dell'Orletta, F.","Fazzone, C.","Miaschi, A.","Venturi, G."],"authors_source":"Alzetta, Chiara; Dell'Orletta, Felice; Fazzone, Chiara; Miaschi, Alessio; Venturi, Giulia","authors_cnr_name":["ALZETTA, CHIARA","DELL'ORLETTA, FELICE","FAZZONE, CHIARA","MIASCHI, ALESSIO","VENTURI, GIULIA"],"authors_cnr_id":["rp12530","rp22811","rp16373","rp12522","rp00732"],"authors_cnr_institute":[],"abstract":"Traditional genre-based approaches for book recommendations face challenges due to the vague definition of genres. To overcome this, we propose a novel task called Book Author Prediction, where we predict the author of a book based on user-generated reviews\u2019 writing style. To this aim, we first introduce the \u2018Literary Voices Corpus\u2019 (LVC), a dataset of Italian book reviews, and use it to train and test machine learning models. Our study contributes valuable insights for developing user-centric systems that recommend leisure readings based on individual readers\u2019 interests and writing styles","keywords":[],"pages":"","url":"https:\/\/ceur-ws.org\/Vol-3596\/paper4.pdf","volume":"","doi":"","editors":[],"editors_source":"","published":"Proceedings of the 9th Italian Conference on Computational Linguistics","publisher":"","issn":"","isbn":"","conference_name":"9th Italian Conference on Computational Linguistics","conference_place":"","conference_date":"","last_updated_cnr":"2025-01-24 23:18:13","last_updated_oai":"2025-01-24 23:18:13","last_updated_www":"0000-00-00 00:00:00"},{"id":1238,"id_source":440157,"institutes":["ILC"],"type":"journal_article","type_order":1,"title":"Linguistically-Based Comparison of Different Approaches to Building Corpora for Text Simplification: A Case Study on Italian","year":2022,"authors":["Brunato, D.","Dell'Orletta, F.","Venturi, G."],"authors_source":"Brunato, Dominique; Dell'Orletta, Felice; Venturi, Giulia","authors_cnr_name":["BRUNATO, DOMINIQUE PIERINA","DELL'ORLETTA, FELICE","VENTURI, GIULIA"],"authors_cnr_id":["rp06836","rp22811","rp00732"],"authors_cnr_institute":[],"abstract":"In this paper, we present an overview of existing parallel corpora for Automatic Text Simplification (ATS) in different languages focusing on the approach adopted for their construction. We make the main distinction between manual and (semi)-automatic approaches in order to investigate in which respect complex and simple texts vary and whether and how the observed modifications may depend on the underlying approach. To this end, we perform a two-level comparison on Italian corpora, since this is the only language, with the exception of English, for which there are large parallel resources derived through the two approaches considered. The first level of comparison accounts for the main types of sentence transformations occurring in the simplification process, the second one examines the results of a linguistic profiling analysis based on Natural Language Processing techniques and carried out on the original and the simple version of the same texts. For both levels of analysis, we chose to focus our discussion mostly on sentence transformations and linguistic characteristics that pertain to the morpho-syntactic and syntactic structure of the sentence","keywords":["linguistic complexity","corpus construction","text simplification"],"pages":"1-19","url":"https:\/\/www.frontiersin.org\/articles\/10.3389\/fpsyg.2022.707630\/full","volume":"13","doi":"10.3389\/fpsyg.2022.707630","editors":[],"editors_source":"","published":"FRONTIERS IN PSYCHOLOGY","publisher":"","issn":"1664-1078","isbn":"","conference_name":"","conference_place":"","conference_date":"","last_updated_cnr":"2025-03-03 07:05:27","last_updated_oai":"2025-03-03 07:05:27","last_updated_www":"0000-00-00 00:00:00"},{"id":244,"id_source":420475,"institutes":["ILC"],"type":"journal_article","type_order":1,"title":"Why is this language complex? Cherry-pick the optimal set of features in multilingual treebanks","year":2022,"authors":["Brunato, D.","Venturi, G."],"authors_source":"Brunato, D; Venturi, G","authors_cnr_name":["BRUNATO, DOMINIQUE PIERINA","VENTURI, GIULIA"],"authors_cnr_id":["rp06836","rp00732"],"authors_cnr_institute":[],"abstract":"This paper investigates linguistic complexity across natural languages from a corpus-based perspective and relies on the assumptions of linguistic profiling as a methodological framework. We focus in particular on the domain of syntactic complexity and analyze the distribution of a set of features taken as proxies of complexity phenomena at the sentence level, which were extracted from 63 treebanks annotated according to the Universal Dependencies formalism. This dataset guarantees that the features considered are modeling the same linguistic phenomena in different treebanks, allowing reliable comparison among languages. We show that our approach is able to identify tendencies of structural proximity between languages not necessarily in line with typologically-supported classification, thus shedding light on new corpus-based findings","keywords":["Linguistic Complexity","Linguistic Profiling","Universal Dependencies"],"pages":"59-72","url":"https:\/\/www.degruyter.com\/document\/doi\/10.1515\/lingvan-2021-0017\/html","volume":"","doi":"10.1515\/lingvan-2021-0017","editors":[],"editors_source":"","published":"LINGUISTICS VANGUARD","publisher":"","issn":"2199-174X","isbn":"","conference_name":"","conference_place":"","conference_date":"","last_updated_cnr":"2025-07-22 01:23:58","last_updated_oai":"2025-07-22 01:23:58","last_updated_www":"0000-00-00 00:00:00"},{"id":180,"id_source":420230,"institutes":["ILC"],"type":"journal_article","type_order":1,"title":"Higher readability of institutional websites drives the correct fruition of the abortion pathway: A cross-sectional study","year":2022,"authors":["Ferrari, A.","Pirrotta, L.","Bonciani, M.","Venturi, G.","Vainieri, M."],"authors_source":"Amerigo Ferrari; Luca Pirrotta; Manila Bonciani; Giulia Venturi; Milena Vainieri","authors_cnr_name":["VENTURI, GIULIA"],"authors_cnr_id":["rp00732"],"authors_cnr_institute":[],"abstract":"In Italy, abortion services are public: therefore, health Institutions should provide clear and easily readable web-based information. We aimed to 1) assess variation in abortion services utilisation; 2) analyse the readability of institutional websites informing on induced abortion; 3) explore whether easier-to-read institutional websites influenced the correct fruition of abortion services. We identified from the 2021 administrative databases of Tuscany all women having an abortion, and-among them-women having an abortion with the certification provided by family counselling centres, following the pathway established by law. We assessed variation in total and certified abortion rates by computing the Systematic Component of Variation. We analysed the readability of the Tuscan health authorities' websites using the readability assessment tool READ-IT. We explored how institutional website readability influenced the odds of having certified abortions by running multilevel logistic models, considering health authorities as the highest-level variables. We observed high variation in the correct utilization of the abortion pathway in terms of certified abortion rates. The READ-IT scores showed that the most readable text was from the Florence Teaching Hospital website. Multilevel models revealed that higher READ-IT scores, corresponding to more difficult texts, resulted in lower odds of certified abortions. Large variation in the proper fruition of abortion pathways occurs in Tuscany, and such variation may depend on readability of institutional websites informing on induced abortion. Therefore, health Institutions should monitor and improve the readability of their websites to ensure proper and more equitable access to abortion","keywords":["abortion services","readability assessment"],"pages":"1-13","url":"https:\/\/journals.plos.org\/plosone\/article?id=10.1371\/journal.pone.0277342","volume":"17 (11)","doi":"10.1371\/journal.pone.0277342","editors":[],"editors_source":"","published":"PLOS ONE","publisher":"","issn":"1932-6203","isbn":"","conference_name":"","conference_place":"","conference_date":"","last_updated_cnr":"2024-07-09 11:30:19","last_updated_oai":"2024-07-09 11:30:19","last_updated_www":"0000-00-00 00:00:00"},{"id":1383,"id_source":417257,"institutes":["ILC"],"type":"journal_article","type_order":1,"title":"On Robustness and Sensitivity of a Neural Language Model: A Case Study on Italian L1 Learner Errors","year":2022,"authors":["Miaschi, A.","Brunato, D. P.","Dell'Orletta, F.","Venturi, G."],"authors_source":"Miaschi, Alessio; Alessio, ; Brunato, DOMINIQUE PIERINA; Dominique, ; Dell'Orletta, Felice; Felice, ; Venturi, Giulia; Giulia,","authors_cnr_name":["MIASCHI, ALESSIO","BRUNATO, DOMINIQUE PIERINA","DELL'ORLETTA, FELICE","VENTURI, GIULIA"],"authors_cnr_id":["rp12522","rp06836","rp22811","rp00732"],"authors_cnr_institute":[],"abstract":"In this paper, we propose a comprehensive linguistic study aimed at assessing the implicit behavior of one of the most prominent Neural Language Models (NLM) based on Transformer architectures, BERT (Devlin et al., 2019), when dealing with a particular source of noisy data, namely essays written by L1 Italian learners containing a variety of errors targeting grammar, orthography and lexicon. Differently from previous works, we focus on the pre-training stage and we devise two complementary evaluation tasks aimed at assessing the impact of errors on sentence-level inner representations in terms of semantic robustness and linguistic sensitivity. While the first evaluation perspective is meant to probe the model's ability to encode the semantic similarity between sentences also in the presence of errors, the second type of probing task evaluates the influence of errors on BERT's implicit knowledge of a set of raw and morpho-syntactic properties of a sentence. Our experiments show that BERT's ability to compute sentence similarity and to correctly encode multi-leveled linguistic information of a sentence are differently modulated by the category of errors and that the error hierarchies in terms of robustness and sensitivity change across layer-wise representations","keywords":["Natural Language Processing","Neural Language Model","Interpretability"],"pages":"426-438","url":"https:\/\/doi.org\/10.1109\/TASLP.2022.3226333","volume":"31","doi":"10.1109\/TASLP.2022.3226333","editors":[],"editors_source":"","published":"IEEE\/ACM TRANSACTIONS ON AUDIO, SPEECH, AND LANGUAGE PROCESSING","publisher":"","issn":"2329-9290","isbn":"","conference_name":"","conference_place":"","conference_date":"","last_updated_cnr":"2025-05-07 00:38:54","last_updated_oai":"2025-05-07 00:38:54","last_updated_www":"0000-00-00 00:00:00"},{"id":1747,"id_source":443057,"institutes":["ILC"],"type":"journal_article","type_order":1,"title":"Probing Linguistic Knowledge in Italian Neural Language Models across Language Varieties","year":2022,"authors":["Miaschi, A.","Sarti, G.","Brunato, D. P.","Dell'Orletta, F.","Venturi, G."],"authors_source":"Miaschi, Alessio; Miaschi, Alessio; Sarti, ; Gabriele, ; Brunato, DOMINIQUE PIERINA; Brunato, DOMINIQUE PIERINA; Dell'Orletta, Felice; Dell'Orletta, Felice; Venturi, Giulia; Venturi, Giulia","authors_cnr_name":["MIASCHI, ALESSIO","MIASCHI, ALESSIO","BRUNATO, DOMINIQUE PIERINA","BRUNATO, DOMINIQUE PIERINA","DELL'ORLETTA, FELICE","DELL'ORLETTA, FELICE","VENTURI, GIULIA","VENTURI, GIULIA"],"authors_cnr_id":["rp12522","rp12522","rp06836","rp06836","rp22811","rp22811","rp00732","rp00732"],"authors_cnr_institute":[],"abstract":"In this paper, we present an in-depth investigation of the linguistic knowledge encoded by the transformer models currently available for the Italian language. In particular, we investigate how the complexity of two different architectures of probing models affects the performance of the Transformers in encoding a wide spectrum of linguistic features. Moreover, we explore how this implicit knowledge varies according to different textual genres and language varieties","keywords":["Neural Language Models","Interpretability","Language Varieties"],"pages":"25-44","url":"http:\/\/www.aaccademia.it\/ita\/scheda-libro?aaref=1518","volume":"","doi":"10.4000\/ijcol.965","editors":[],"editors_source":"","published":"IJCOL","publisher":"","issn":"2499-4553","isbn":"","conference_name":"","conference_place":"","conference_date":"","last_updated_cnr":"2025-02-05 22:52:50","last_updated_oai":"2025-02-05 22:52:50","last_updated_www":"0000-00-00 00:00:00"},{"id":341,"id_source":445825,"institutes":["ILC"],"type":"journal_article","type_order":1,"title":"COVID-19 vaccinations: An overview of the Italian national health system's online communication from a citizen perspective","year":2022,"authors":["Pirrotta, L.","Guidotti, E.","Tramontani, C.","Bignardelli, E.","Venturi, G.","De Rosis, S."],"authors_source":"Pirrotta, L; Guidotti, E; Tramontani, C; Bignardelli, E; Venturi, Giulia; De Rosis, S","authors_cnr_name":["VENTURI, GIULIA"],"authors_cnr_id":["rp00732"],"authors_cnr_institute":[],"abstract":"COVID-19 vaccine hesitancy is still widespread. During the pandemic, the internet has been the preferred channel for health-related information, especially for less-educated citizens who tend to be the most hesitant about vaccination. A well-structured web communication strategy could help both to overcome vaccine hesitancy and to ensure equity in healthcare service access. This study investigated how the various regional and local health authorities in Italy used their institutional websites to inform users about COVID-19 vaccinations between March and April 2021. We browsed 129 institutional websites, checking the availability, quality and quantity, actionability and readability of information using a literature-based common grid. Descriptive statistics and statistical tests were performed. The online public dissemination of COVID-19 vaccination information in Italy was fragmented, both across and within regions. The side effects of vaccinations, were often not reported on the websites, thus missing an opportunity to enhance vaccination uptake. More focus should also be placed on readability, since readability indexes showed that they were difficult to understand. Our research revealed that several actions could be implemented to enhance online communication on COVID-19 vaccination. For instance, simplifying texts can make them more understandable and the information reported actionable","keywords":["Vaccinationa Communication","Readability Assessment","Online Information","Covid-19"],"pages":"970-979","url":"https:\/\/www.sciencedirect.com\/science\/article\/pii\/S0168851022002184","volume":"10 (126)","doi":"10.1016\/j.healthpol.2022.08.001","editors":[],"editors_source":"","published":"HEALTH POLICY","publisher":"","issn":"0168-8510","isbn":"","conference_name":"","conference_place":"","conference_date":"","last_updated_cnr":"2025-02-14 07:06:51","last_updated_oai":"2025-02-14 07:06:51","last_updated_www":"0000-00-00 00:00:00"},{"id":182,"id_source":440167,"institutes":["ILC"],"type":"book","type_order":3,"title":"La fede dichiarata. Un'analisi linguistico-computazionale","year":2022,"authors":["Venturi, G.","Cimino, A.","Dell'Orletta, F."],"authors_source":"Venturi, Giulia; Cimino, Andrea; Dell'Orletta, Felice","authors_cnr_name":["VENTURI, GIULIA","DELL'ORLETTA, FELICE"],"authors_cnr_id":["rp00732","rp22811"],"authors_cnr_institute":[],"abstract":"Il volume indaga l'apporto di tecnologie basate sul Natural Language Processing (NLP) all'analisi di un corpus di trascrizioni di 164 interviste orali raccolte durante la ricerca 2017 sulla \"Religiosit\u00e0 in Italia\". Gli autori illustrano metodologie e strumenti che permettono di trasformare l'informazione implicitamente contenuta nelle interviste in informazione esplicitamente strutturata. Il risultato finale di questo processo interpretativo spazia dall'acquisizione di conoscenze lessicali e terminologiche complesse alla loro organizzazione in strutture proto-concettuali, fino ad arrivare alla qualificazione dell'atteggiamento con il quale l'intervistato si esprime. Il lettore viene accompagnato a scoprire quale sia il valore aggiunto delle analisi basate su NLP e quali nuovi orizzonti di ricerca siano aperti da queste analisi","keywords":["Knowledge Extraction","Knowledge Organization"],"pages":"1-181","url":"https:\/\/iris.cnr.it\/handle\/20.500.14243\/440167","volume":"","doi":"","editors":[],"editors_source":"","published":"","publisher":"Franco Angeli Editore (Milano, ITA)","issn":"","isbn":"978-88-351-2146-6","conference_name":"","conference_place":"Milano","conference_date":"","last_updated_cnr":"2024-12-18 15:58:00","last_updated_oai":"2024-12-18 15:58:00","last_updated_www":"0000-00-00 00:00:00"},{"id":248,"id_source":446358,"institutes":["ILC","IGSG"],"type":"conference_article","type_order":7,"title":"Making Italian Parliamentary Records Machine-Actionable: the Construction of the ParlaMint-IT corpus","year":2022,"authors":["Agnoloni, T.","Bartolini, R.","Frontini, F.","Montemagni, S.","Marchetti, C.","Quochi, V.","Ruisi, M.","Venturi, G."],"authors_source":"Agnoloni, Tommaso; Bartolini, Roberto; Frontini, Francesca; Montemagni, Simonetta; Marchetti, Carlo; Quochi, Valeria; Ruisi, Manuela; Venturi, Giulia","authors_cnr_name":["AGNOLONI, TOMMASO","BARTOLINI, ROBERTO","FRONTINI, FRANCESCA","MONTEMAGNI, SIMONETTA","QUOCHI, VALERIA","VENTURI, GIULIA"],"authors_cnr_id":["rp21506","rp00239","rp02790","rp16780","rp13283","rp00732"],"authors_cnr_institute":[],"abstract":"This paper describes the process of acquisition, cleaning, interpretation, coding and linguistic annotation of a collection of parliamentary debates from the Senate of the Italian Republic covering the COVID-19 pandemic emergency period and a former period for reference and comparison according to the CLARIN ParlaMint prescriptions. The corpus contains 1199 sessions and 79, 373 speeches for a total of about 31 million words, and was encoded according to the ParlaCLARIN TEI XML format. It includes extensive metadata about the speakers, sessions, political parties and parliamentary groups. As required by the ParlaMint initiative, the corpus was also linguistically annotated for sentences, tokens, POS tags, lemmas and dependency syntax according to the universal dependencies guidelines. Named entity annotation and classification is also included. All linguistic annotation was performed automatically using state-of-the-art NLP technology with no manual revision. The Italian dataset is freely available as part of the larger ParlaMint 2. 1 corpus deposited and archived in CLARIN repository together with all other national corpora. It is also available for direct analysis and inspection via various CLARIN services and has already been used both for research and educational purposes","keywords":["parliamentary debates","CLARIN ParlaMint","corpus creation","corpus annotation"],"pages":"117-124","url":"https:\/\/aclanthology.org\/2022.parlaclarin-1.17\/","volume":"","doi":"","editors":[],"editors_source":"","published":"Proceedings of The Workshop ParlaCLARIN III within the 13th Language Resources and Evaluation Conference","publisher":"European Language Resources Association ELRA (Paris, FRA)","issn":"","isbn":"979-10-95546-85-6","conference_name":"Workshop ParlaCLARIN III within the 13th Language Resources and Evaluation Conference","conference_place":"Paris","conference_date":"","last_updated_cnr":"2025-05-22 00:35:15","last_updated_oai":"2025-05-22 00:35:15","last_updated_www":"0000-00-00 00:00:00"},{"id":194,"id_source":448002,"institutes":["ILC"],"type":"conference_article","type_order":7,"title":"SemEval-2022 Task 3: PreTENS-Evaluating Neural Networks on Presuppositional Semantic Knowledge","year":2022,"authors":["Zamparelli, R.","A Chowdhury, S.","Brunato, D.","Chesi, C.","Dell'Orletta, F.","Hasan, A.","Venturi, G."],"authors_source":"Zamparelli, Roberto; A Chowdhury, Shammur; Brunato, Dominique; Chesi, Cristiano; Dell'Orletta, Felice; Hasan, Arid; Venturi, Giulia","authors_cnr_name":["BRUNATO, DOMINIQUE PIERINA","DELL'ORLETTA, FELICE","VENTURI, GIULIA"],"authors_cnr_id":["rp06836","rp22811","rp00732"],"authors_cnr_institute":[],"abstract":"We report the results of the SemEval 2022 Task 3, PreTENS, on evaluation the acceptability of simple sentences containing constructions whose two arguments are presupposed to be or not to be in an ordered taxonomic relation. The task featured two sub-tasks articulated as: (i) binary prediction task and (ii) regression task, predicting the acceptability in a continuous scale. The sentences were artificially generated in three languages (English, Italian and French). 21 systems, with 8 system papers were submitted for the task, all based on various types of fine-tuned transformer systems, often with ensemble methods and various data augmentation techniques. The best systemsreached an F1-macro score of 94. 49 (sub-task1) and a Spearman correlation coefficient of 0. 80 (sub-task2), with interesting variations in specific constructions and\/or languages","keywords":["Neural Networks","Presuppositional Knowledge","Evaluation"],"pages":"228-238","url":"https:\/\/aclanthology.org\/2022.semeval-1.29.pdf","volume":"","doi":"","editors":[],"editors_source":"","published":"16th International Workshop on Semantic Evaluation (SemEval-2022)","publisher":"","issn":"","isbn":"","conference_name":"16th International Workshop on Semantic Evaluation (SemEval-2022)","conference_place":"","conference_date":"","last_updated_cnr":"2024-12-18 04:29:38","last_updated_oai":"2024-12-18 04:29:38","last_updated_www":"0000-00-00 00:00:00"},{"id":881,"id_source":446048,"institutes":["ILC"],"type":"conference_article","type_order":7,"title":"Probing tasks under pressure","year":2021,"authors":["Miaschi, A.","Alzetta, C.","Brunato, D.","Dell'Orletta, F.","Venturi, G."],"authors_source":"Miaschi A.; Alzetta C.; Brunato D.; Dell'Orletta F.; Venturi G.","authors_cnr_name":["MIASCHI, ALESSIO","ALZETTA, CHIARA","BRUNATO, DOMINIQUE PIERINA","DELL'ORLETTA, FELICE","VENTURI, GIULIA"],"authors_cnr_id":["rp12522","rp12530","rp06836","rp22811","rp00732"],"authors_cnr_institute":[],"abstract":"Probing tasks are frequently used to evaluate whether the representations of Neural Language Models (NLMs) encode linguistic information. However, it is still questioned if probing classification tasks really enable such investigation or they simply hint for surface patterns in the data. We present a method to investigate this question by comparing the accuracies of a set of probing tasks on gold and automatically generated control datasets. Our results suggest that probing tasks can be used as reliable diagnostic methods to investigate the linguistic information encoded in NLMs representations","keywords":["Neural Language Models","Linguistic probing","Treebanks"],"pages":"1-7","url":"http:\/\/ceur-ws.org\/Vol-3033\/paper29.pdf","volume":"3033","doi":"","editors":[],"editors_source":"","published":"","publisher":"","issn":"","isbn":"","conference_name":"8th Italian Conference on Computational Linguistics (CLIC-it 2021)","conference_place":"","conference_date":"","last_updated_cnr":"2024-12-22 03:17:12","last_updated_oai":"2024-12-22 03:17:12","last_updated_www":"0000-00-00 00:00:00"},{"id":1549,"id_source":400474,"institutes":["ILC"],"type":"conference_article","type_order":7,"title":"What Makes My Model Perplexed? A Linguistic Investigation on Neural Language Models Perplexity","year":2021,"authors":["Miaschi, A.","Brunato, D. P.","Dell'Orletta, F.","Venturi, G."],"authors_source":"Miaschi, Alessio; Miaschi, Alessio; Brunato, DOMINIQUE PIERINA; Brunato, DOMINIQUE PIERINA; Dell'Orletta, Felice; Dell'Orletta, Felice; Venturi, Giulia; Venturi, Giulia","authors_cnr_name":["MIASCHI, ALESSIO","MIASCHI, ALESSIO","BRUNATO, DOMINIQUE PIERINA","BRUNATO, DOMINIQUE PIERINA","DELL'ORLETTA, FELICE","DELL'ORLETTA, FELICE","VENTURI, GIULIA","VENTURI, GIULIA"],"authors_cnr_id":["rp12522","rp12522","rp06836","rp06836","rp22811","rp22811","rp00732","rp00732"],"authors_cnr_institute":[],"abstract":"This paper presents an investigation aimed at studying how the linguistic structure of a sentence affects the perplexity of two of the most popular Neural Language Models (NLMs), BERT and GPT-2. We first compare the sentence-level likelihood computed with BERT and the GPT-2's perplexity showing that the two metrics are correlated. In addition, we exploit linguistic features capturing a wide set of morpho-syntactic and syntactic phenomena showing how they contribute to predict the perplexity of the two NLMs","keywords":["nlp","interpretability","deep learning"],"pages":"40-47","url":"https:\/\/www.aclweb.org\/anthology\/2021.deelio-1.5","volume":"","doi":"","editors":[],"editors_source":"","published":"Proceedings of the 2nd Workshop on Knowledge Extraction and Integrationfor Deep Learning Architectures","publisher":"","issn":"","isbn":"978-1-954085-30-5","conference_name":"2nd Workshop on Knowledge Extraction and Integrationfor Deep Learning Architectures","conference_place":"","conference_date":"","last_updated_cnr":"2024-12-20 02:05:59","last_updated_oai":"2024-12-20 02:05:59","last_updated_www":"0000-00-00 00:00:00"},{"id":383,"id_source":446076,"institutes":["IGSG","ILC"],"type":"misc","type_order":11,"title":"Linguistically annotated multilingual comparable corpora of parliamentary debates ParlaMint. ana 2. 1","year":2021,"authors":["Erjavec, T.","Ogrodniczuk, M.","Osenova, P.","Ljubei, N.","Simov, K.","Grigorova, V.","Rudolf, M.","Panur, A.","Kopp, M.","Barkarson, S.","Steingr\u00edmsson, S.","Van Der Pol, H.","Depoorter, G.","De Does, J.","Jongejan, B.","Haltrup Hansen, D.","Navarretta, C.","Calzada P\u00e9rez, M.","D De Macedo, L.","Van Heusden, R.","Marx, M.","\u00c7\u00f6ltekin, \u00c7.","Coole, M.","Agnoloni, T.","Frontini, F.","Montemagni, S.","Quochi, V.","Venturi, G.","Ruisi, M.","Marchetti, C.","Battistoni, R.","Sebk, M.","Ring, O.","Daris, R.","Utka, A.","Petkeviius, M.","Briedien\u00e9, M.","Krilaviius, T.","Morkeviius, V.","Bartolini, R.","Cimino, A.","Diwersy, S.","Luxardo, G.","Rayson, P."],"authors_source":"Erjavec, Toma; Ogrodniczuk, Maciej; Osenova, Petya; Ljubei, Nikola; Simov, Kiril; Grigorova, Vladislava; Rudolf, Micha; Panur, Andrej; Kopp, Maty\u00e1; Barkarson, Starka\u00f0ur; Steingr\u00edmsson, Stein\u00feor; van der Pol, Henk; Depoorter, Griet; de Does, Jesse; Jongejan, Bart; Haltrup Hansen, Dorte; Navarretta, Costanza; Calzada P\u00e9rez, Mar\u00eda; D de Macedo, Luciana; van Heusden, Ruben; Marx, Maarten; \u00c7\u00f6ltekin, \u00c7ar; Coole, Matthew; Agnoloni, Tommaso; Frontini, Francesca; Montemagni, Simonetta; Quochi, Valeria; Venturi, Giulia; Ruisi, Manuela; Marchetti, Carlo; Battistoni, Roberto; Sebk, Mikl\u00f3s; Ring, Orsolya; Daris, Roberts; Utka, Andrius; Petkeviius, Mindaugas; Briedien\u00e9, Monika; Krilaviius, Tomas; Morkeviius, Vaidas; Bartolini, Roberto; Cimino, Andrea; Diwersy, Sascha; Luxardo, Giancarlo; Rayson, Paul","authors_cnr_name":["AGNOLONI, TOMMASO","FRONTINI, FRANCESCA","MONTEMAGNI, SIMONETTA","QUOCHI, VALERIA","VENTURI, GIULIA","BARTOLINI, ROBERTO","CIMINO, ANDREA"],"authors_cnr_id":["rp21506","rp02790","rp16780","rp13283","rp00732","rp00239","rp05770"],"authors_cnr_institute":[],"abstract":"ParlaMint 2. 1 is a multilingual set of 17 comparable corpora containing parliamentary debates mostly starting in 2015 and extending to mid-2020, with each corpus being about 20 million words in size. The sessions in the corpora are marked as belonging to the COVID-19 period (from November 1st 2019), or being \"reference\" (before that date). The corpora have extensive metadata, including aspects of the parliament; the speakers (name, gender, MP status, party affiliation, party coalition\/opposition); are structured into time-stamped terms, sessions and meetings; with speeches being marked by the speaker and their role (e. g. chair, regular speaker). The speeches also contain marked-up transcriber comments, such as gaps in the transcription, interruptions, applause, etc. Note that some corpora have further information, e. g. the year of birth of the speakers, links to their Wikipedia articles, their membership in various committees, etc. The corpora are encoded according to the Parla-CLARIN TEI recommendation (https: \/\/clarin-eric. github. io\/parla-clarin\/), but have been validated against the compatible, but much stricter ParlaMint schemas. This entry contains the linguistically marked-up version of the corpus, while the text version is available at http: \/\/hdl. handle. net\/11356\/1432. The ParlaMint. ana linguistic annotation includes tokenization, sentence segmentation, lemmatisation, Universal Dependencies part-of-speech, morphological features, and syntactic dependencies, and the 4-class CoNLL-2003 named entities. Some corpora also have further linguistic annotations, such as PoS tagging or named entities according to language-specific schemes, with their corpus TEI headers giving further details on the annotation vocabularies and tools","keywords":["covid-19","ParlaCLARIN","CLARIN","linguistic annotation","pos-tagging","Named Entity Recognition","linguistic dependency annotation","UD","dibattiti parlamentari","parlamenti","discorso politico"],"pages":"","url":"https:\/\/iris.cnr.it\/handle\/20.500.14243\/446076","volume":"","doi":"","editors":[],"editors_source":"","published":"","publisher":"","issn":"","isbn":"","conference_name":"","conference_place":"","conference_date":"","last_updated_cnr":"2025-03-07 22:39:26","last_updated_oai":"2025-03-07 22:39:26","last_updated_www":"0000-00-00 00:00:00"},{"id":2175,"id_source":446080,"institutes":["IGSG","ILC"],"type":"misc","type_order":11,"title":"Multilingual comparable corpora of parliamentary debates ParlaMint 2. 1","year":2021,"authors":["Erjavec, T.","Ogrodniczuk, M.","Osenova, P.","Ljubei, N.","Simov, K.","Grigorova, V.","Rudolf, M.","Panur, A.","Kopp, M.","Barkarson, S.","Steingr\u00edmsson, S.","Van Der Pol, H.","Depoorter, G.","De Does, J.","Jongejan, B.","Haltrup Hansen, D.","Navarretta, C.","Calzada P\u00e9rez, M.","D De Macedo, L.","Van Heusden, R.","Marx, M.","\u00c7\u00f6ltekin, \u00c7.","Coole, M.","Agnoloni, T.","Frontini, F.","Montemagni, S.","Quochi, V.","Venturi, G.","Ruisi, M.","Marchetti, C.","Battistoni, R.","Sebk, M.","Ring, O.","Daris, R.","Utka, A.","Petkeviius, M.","Briedien\u00e9, M.","Krilaviius, T.","Morkeviius, V.","Bartolini, R.","Cimino, A.","Diwersy, S.","Luxardo, G.","Rayson, P."],"authors_source":"Erjavec, Toma; Ogrodniczuk, Maciej; Osenova, Petya; Ljubei, Nikola; Simov, Kiril; Grigorova, Vladislava; Rudolf, Micha; Panur, Andrej; Kopp, Maty\u00e1; Barkarson, Starka\u00f0ur; Steingr\u00edmsson, Stein\u00feor; van der Pol, Henk; Depoorter, Griet; de Does, Jesse; Jongejan, Bart; Haltrup Hansen, Dorte; Navarretta, Costanza; Calzada P\u00e9rez, Mar\u00eda; D de Macedo, Luciana; van Heusden, Ruben; Marx, Maarten; \u00c7\u00f6ltekin, \u00c7ar; Coole, Matthew; Agnoloni, Tommaso; Frontini, Francesca; Montemagni, Simonetta; Quochi, Valeria; Venturi, Giulia; Ruisi, Manuela; Marchetti, Carlo; Battistoni, Roberto; Sebk, Mikl\u00f3s; Ring, Orsolya; Daris, Roberts; Utka, Andrius; Petkeviius, Mindaugas; Briedien\u00e9, Monika; Krilaviius, Tomas; Morkeviius, Vaidas; Bartolini, Roberto; Cimino, Andrea; Diwersy, Sascha; Luxardo, Giancarlo; Rayson, Paul","authors_cnr_name":["AGNOLONI, TOMMASO","FRONTINI, FRANCESCA","MONTEMAGNI, SIMONETTA","QUOCHI, VALERIA","VENTURI, GIULIA","BARTOLINI, ROBERTO"],"authors_cnr_id":["rp21506","rp02790","rp16780","rp13283","rp00732","rp00239"],"authors_cnr_institute":[],"abstract":"ParlaMint 2. 1 is a multilingual set of 17 comparable corpora containing parliamentary debates mostly starting in 2015 and extending to mid-2020, with each corpus being about 20 million words in size. The sessions in the corpora are marked as belonging to the COVID-19 period (after November 1st 2019), or being \"reference\" (before that date). The corpora have extensive metadata, including aspects of the parliament; the speakers (name, gender, MP status, party affiliation, party coalition\/opposition); are structured into time-stamped terms, sessions and meetings; with speeches being marked by the speaker and their role (e. g. chair, regular speaker). The speeches also contain marked-up transcriber comments, such as gaps in the transcription, interruptions, applause, etc. Note that some corpora have further information, e. g. the year of birth of the speakers, links to their Wikipedia articles, their membership in various committees, etc. The corpora are encoded according to the Parla-CLARIN TEI recommendation (https: \/\/clarin-eric. github. io\/parla-clarin\/), but have been validated against the compatible, but much stricter ParlaMint schemas. This entry contains the ParlaMint TEI-encoded corpora with the derived plain text version of the corpus along with TSV metadata on the speeches. Also included is the 2. 0 release of the data and scripts available at the GitHub repository of the ParlaMint project. Note that there also exists the linguistically marked-up version of the corpus, which is available at http: \/\/hdl. handle. net\/11356\/1431","keywords":[],"pages":"","url":"https:\/\/iris.cnr.it\/handle\/20.500.14243\/446080","volume":"","doi":"","editors":[],"editors_source":"","published":"","publisher":"","issn":"","isbn":"","conference_name":"","conference_place":"","conference_date":"","last_updated_cnr":"2025-03-07 06:32:49","last_updated_oai":"0000-00-00 00:00:00","last_updated_www":"0000-00-00 00:00:00"},{"id":1812,"id_source":446043,"institutes":["ILC"],"type":"journal_article","type_order":1,"title":"Linguistically-driven Selection of Difficult-to-Parse Dependency Structures","year":2020,"authors":["Alzetta, C.","Dell'Orletta, F.","Montemagni, S.","Venturi, G."],"authors_source":"Alzetta, Chiara; Dell'Orletta, Felice; Montemagni, Simonetta; Venturi, Giulia","authors_cnr_name":["ALZETTA, CHIARA","DELL'ORLETTA, FELICE","MONTEMAGNI, SIMONETTA","VENTURI, GIULIA"],"authors_cnr_id":["rp12530","rp22811","rp16780","rp00732"],"authors_cnr_institute":[],"abstract":"The paper illustrates a novel methodology meeting a twofold goal, namely quantifying the reliability of automatically generated dependency relations without using gold data on the one hand, and identifying which are the linguistic constructions negatively affecting the parser performance on the other hand. These represent objectives typically investigated in different lines of research, with different methods and techniques. Our methodology, at the crossroads of these perspectives, allows not only to quantify the parsing reliability of individual dependency types but also to identify and weight the contextual properties making relation instances more or less difficult to parse. The proposed methodology was tested in two different and complementary experiments, aimed at assessing the degree of parsing difficulty across (a) different dependency relation types, and (b) different instances of the same relation. The results show that the proposed methodology is able to identify difficult-to-parse dependency relations without relying on gold data and by taking into account a variety of intertwined linguistic factors. These findings pave the way to novel applications of the methodology, both in the direction of defining new evaluation metrics based purely on automatically parsed data and towards the automatic creation of challenge sets","keywords":["Linguistic Complexity","Syntactic Parsing","Evaluation metrics"],"pages":"37-60","url":"https:\/\/journals.openedition.org\/ijcol\/719","volume":"6 (2)","doi":"10.4000\/ijcol.719","editors":[],"editors_source":"","published":"IJCOL","publisher":"","issn":"2499-4553","isbn":"","conference_name":"","conference_place":"","conference_date":"","last_updated_cnr":"2026-03-04 01:28:15","last_updated_oai":"2026-03-04 01:28:15","last_updated_www":"0000-00-00 00:00:00"},{"id":1804,"id_source":384935,"institutes":["ILC"],"type":"journal_article","type_order":1,"title":"Waiting time information in the Italian NHS: A citizen perspective","year":2020,"authors":["De Rosis, S.","Guidotti, E.","Zuccarino, S.","Venturi, G.","Ferre, F."],"authors_source":"De Rosis, Sabina; Guidotti, Elisa; Zuccarino, Sara; Venturi, Giulia; Ferre, Francesca","authors_cnr_name":["VENTURI, GIULIA"],"authors_cnr_id":["rp00732"],"authors_cnr_institute":[],"abstract":"Public involvement in the management and communication of waiting times is known to support initiatives to reduce waiting times, as well as increase fairness and promote transparency and accountability. In order to improve transparency and communication to citizens, Italy recently updated the National Regulatory Plan for Waiting Lists (2019-2021), which calls for the disclosure of waiting time information on healthcare provider webpages. This study analyses waiting time information for outpatient visits and digital services available on the institutional website pages of 144 public healthcare organisations in nine regions and two autonomous provinces of Italy. Web pages were analysed both in terms of the available information\/services, using a grid, and in terms of the quality of the text using an advanced readability assessment tool (READ-IT). This information was complemented and validated by regional healthcare key informants during research-specific workshops. Waiting time information disclosure, digital services and text readability varied both within and between the regional healthcare systems and organisations. The types and characteristics of waiting time information and statistics vary considerably with a negative impact on their use for benchmarking and their readability and usability for booking purposes. Overall, communication weaknesses due to low harmonization and clarity of information can undermine efforts in effectively informing and involving the public through online waiting time data disclosure. (C) 2020 The Author(s). Published by Elsevier B. V","keywords":["Waiting times","Healthcare","Online information","Readability","Italy"],"pages":"796-804","url":"https:\/\/www.sciencedirect.com\/science\/article\/pii\/S0168851020301111?via%3Dihub","volume":"124 (8)","doi":"10.1016\/j.healthpol.2020.05.012","editors":[],"editors_source":"","published":"HEALTH POLICY","publisher":"","issn":"0168-8510","isbn":"","conference_name":"","conference_place":"","conference_date":"","last_updated_cnr":"2025-04-03 00:59:09","last_updated_oai":"2025-04-03 00:59:09","last_updated_www":"0000-00-00 00:00:00"},{"id":1032,"id_source":426118,"institutes":["ILC"],"type":"journal_article","type_order":1,"title":"Metodi e Tecniche di Trattamento Automatico della Lingua per l'Estrazione di Conoscenza dalla Documentazione Scolastica","year":2020,"authors":["Venturi, G.","Dell'Orletta, F.","Montemagni, S.","Morini E, S. M."],"authors_source":"Venturi, G; Dell'Orletta, F; Montemagni, S; Morini E, e Sagri MT","authors_cnr_name":["VENTURI, GIULIA","DELL'ORLETTA, FELICE","MONTEMAGNI, SIMONETTA"],"authors_cnr_id":["rp00732","rp22811","rp16780"],"authors_cnr_institute":[],"abstract":"Il contributo riguarda la creazione di un sistema integrato di \"knowledge management\", per la gestione e condivisione della conoscenza prodotta e utilizzata dalla scuola","keywords":["Estrazione di informazione","Documenti scolastici","Indicizzazione","Terminology extraction"],"pages":"49-68","url":"https:\/\/iris.cnr.it\/handle\/20.500.14243\/426118","volume":"2","doi":"10.3280\/CAD2020-002005","editors":[],"editors_source":"","published":"CADMO","publisher":"","issn":"1122-5165","isbn":"","conference_name":"","conference_place":"","conference_date":"","last_updated_cnr":"2025-03-07 06:28:30","last_updated_oai":"2025-03-07 06:28:30","last_updated_www":"0000-00-00 00:00:00"},{"id":369,"id_source":426114,"institutes":["ILC"],"type":"journal_article","type_order":1,"title":"Verba et Acta. Un esperimento per promuovere l'evoluzione delle compe-tenze linguistiche degli studenti degli istituti professionali","year":2020,"authors":["Vertecchi, B.","Agrusti, F.","Dell'Orletta, F.","Montemagni, S.","Venturi, G."],"authors_source":"Vertecchi, Benedetto; Agrusti, Francesco; Dell'Orletta, Felice; Montemagni, Simonetta; Venturi, Giulia","authors_cnr_name":["DELL'ORLETTA, FELICE","MONTEMAGNI, SIMONETTA","VENTURI, GIULIA"],"authors_cnr_id":["rp22811","rp16780","rp00732"],"authors_cnr_institute":[],"abstract":"Ricerche in corso. Verba et Acta. Un esperimento per promuovere l'evoluzione delle competenze linguistiche degli studenti degli istituti professionali","keywords":["Evoluzione competenze linguistiche","Annotazione linguistica","Previsione dello sviluppo delle competenze di scrittura"],"pages":"109-117","url":"https:\/\/iris.cnr.it\/handle\/20.500.14243\/426114","volume":"(1)","doi":"10.3280\/CAD2020-001008","editors":[],"editors_source":"","published":"CADMO","publisher":"","issn":"1122-5165","isbn":"","conference_name":"","conference_place":"","conference_date":"","last_updated_cnr":"2025-02-23 00:45:54","last_updated_oai":"2025-02-23 00:45:54","last_updated_www":"0000-00-00 00:00:00"},{"id":1201,"id_source":423610,"institutes":["ILC"],"type":"conference_article","type_order":7,"title":"Quantitative linguistic investigations across universal dependencies treebanks","year":2020,"authors":["Alzetta, C.","Dell'Orletta, F.","Montemagni, S.","Osenova, P.","Simov, K.","Venturi, G."],"authors_source":"Alzetta C.; Dell'Orletta F.; Montemagni S.; Osenova P.; Simov K.; Venturi G.","authors_cnr_name":["ALZETTA, CHIARA","DELL'ORLETTA, FELICE","MONTEMAGNI, SIMONETTA","VENTURI, GIULIA"],"authors_cnr_id":["rp12530","rp22811","rp16780","rp00732"],"authors_cnr_institute":[],"abstract":"The paper illustrates a case study aimed at identifying cross-lingual quantitative trends in the distribution of dependency relations in treebanks for typologically different languages. Preliminary results show interesting differences rooted either in language-specific peculiarities or cross-lingual annotation inconsistencies, with a potential impact on different application scenarios","keywords":["Universal Dependencies Treebanks","Cross-linguistic analysis","Typology"],"pages":"1-7","url":"http:\/\/ceur-ws.org\/Vol-2769\/paper_59.pdf","volume":"2769","doi":"","editors":[],"editors_source":"","published":"","publisher":"","issn":"","isbn":"979-12-80136-28-2","conference_name":"7th Italian Conference on Computational Linguistics (CLiC-it)","conference_place":"","conference_date":"","last_updated_cnr":"2024-07-22 15:38:26","last_updated_oai":"2024-07-22 15:38:26","last_updated_www":"0000-00-00 00:00:00"},{"id":1717,"id_source":423611,"institutes":["ILC"],"type":"conference_article","type_order":7,"title":"AcCompl-it @ EVALITA2020: Overview of the acceptability & complexity evaluation task for Italian","year":2020,"authors":["Brunato, D.","Chesi, C.","Dell'Orletta, F.","Montemagni, S.","Venturi, G.","Zamparelli, R."],"authors_source":"Brunato, D; Chesi, C; Dell'Orletta, F; Montemagni, S; Venturi, G; Zamparelli, R","authors_cnr_name":["BRUNATO, DOMINIQUE PIERINA","DELL'ORLETTA, FELICE","MONTEMAGNI, SIMONETTA","VENTURI, GIULIA"],"authors_cnr_id":["rp06836","rp22811","rp16780","rp00732"],"authors_cnr_institute":[],"abstract":"The Acceptability and Complexity evaluation task for Italian (AcCompl-it) was aimed at developing and evaluating methods to classify Italian sentences according to Acceptability and Complexity. It consists of two independent tasks asking participants to predict either the acceptability or the complexity rate (or both) of a given set of sentences previously scored by native speakers on a 1-to-7 points Likert scale. In this paper, we introduce the datasets distributed to the participants, we describe the different approaches of the participating systems and provide a first analysis of the obtained results","keywords":["Shared Task","Linguistic Complexity","Acceptability"],"pages":"1-8","url":"http:\/\/ceur-ws.org\/Vol-2765\/paper163.pdf","volume":"2765","doi":"","editors":[],"editors_source":"","published":"","publisher":"","issn":"","isbn":"","conference_name":"EVALITA '20, Evaluation of NLP and Speech Tools for Italian","conference_place":"","conference_date":"","last_updated_cnr":"2024-06-16 13:08:20","last_updated_oai":"2024-06-16 13:08:20","last_updated_www":"0000-00-00 00:00:00"},{"id":342,"id_source":384930,"institutes":["ILC"],"type":"conference_article","type_order":7,"title":"Profiling-UD: a Tool for Linguistic Profiling of Texts","year":2020,"authors":["Brunato, D.","Cimino, A.","Dell'Orletta, F.","Montemagni, S.","Venturi, G."],"authors_source":"Brunato, Dominique; Cimino, Andrea; Dell'Orletta, Felice; Montemagni, Simonetta; Venturi, Giulia","authors_cnr_name":["BRUNATO, DOMINIQUE PIERINA","CIMINO, ANDREA","DELL'ORLETTA, FELICE","MONTEMAGNI, SIMONETTA","VENTURI, GIULIA"],"authors_cnr_id":["rp06836","rp05770","rp22811","rp16780","rp00732"],"authors_cnr_institute":[],"abstract":"In this paper, we introduce Profiling-UD, a new text analysis tool inspired to the principles of linguistic profiling that can support language variation research from different perspectives. It allows the extraction of more than 130 features, spanning across different levels of linguistic description. Beyond the large number of features that can be monitored, a main novelty of Profiling-UD is that it has been specifically devised to be multilingual since it is based on the Universal Dependencies framework. In the second part of the paper, we demonstrate the effectiveness of these features in a number of theoretical and applicative studies in which they were successfully used for text and author profiling","keywords":["Computational Language Variation Analysis","Linguistic Profiling","Universal Dependencies"],"pages":"7145-7151","url":"http:\/\/www.lrec-conf.org\/proceedings\/lrec2020\/pdf\/2020.lrec-1.883.pdf","volume":"","doi":"","editors":[],"editors_source":"","published":"Proceedings of the 12th Language Resources and Evaluation Conference-LREC 2020","publisher":"European Language Resources Association ELRA (Paris, FRA)","issn":"","isbn":"979-10-95546-34-4","conference_name":"Conference on Language Resources and Evaluation (LREC)","conference_place":"Paris","conference_date":"","last_updated_cnr":"2025-01-11 23:10:10","last_updated_oai":"2025-01-11 23:10:10","last_updated_www":"0000-00-00 00:00:00"},{"id":539,"id_source":384922,"institutes":["ILC"],"type":"conference_article","type_order":7,"title":"Voices of the Great War: A Richly Annotated Corpus of Italian Texts on the First World War","year":2020,"authors":["Lenci, A.","Montemagni, S.","Boschetti, F.","De Felice, I.","Dei Rossi, S.","Dell'Orletta, F.","Di Giorgio, M.","Miliani, M.","C Passaro, L.","Puddu, A.","Venturi, G.","Labanca, N."],"authors_source":"Lenci, Alessandro; Montemagni, Simonetta; Boschetti, Federico; De Felice, Irene; dei Rossi, Stefano; Dell'Orletta, Felice; Di Giorgio, Michele; Miliani, Martina; C Passaro, Lucia; Puddu, Angelica; Venturi, Giulia; Labanca, Nicola","authors_cnr_name":["MONTEMAGNI, SIMONETTA","BOSCHETTI, FEDERICO","DELL'ORLETTA, FELICE","VENTURI, GIULIA"],"authors_cnr_id":["rp16780","rp04876","rp22811","rp00732"],"authors_cnr_institute":[],"abstract":"Voci della Grande Guerra (\"Voices of the Great War\") is the first large corpus of Italian historical texts dating back to the period of First World War. This corpus differs from other existing resources in several respects. First, from the linguistic point of view it gives account of the wide range of varieties in which Italian was articulated in that period, namely from a diastratic (educated vs. uneducated writers), diaphasic (low\/informal vs. high\/formal registers) and diatopic (regional varieties, dialects) points of view. From the historical perspective, through a collection of texts belonging to different genres it represents different views on the war and the various styles of narrating war events and experiences. The final corpus is balanced along various dimensions, corresponding to the textual genre, the language variety used, the author type and the typology of conveyed contents. The corpus is annotated with lemmas, part-of-speech, terminology, and named entities. Significant corpus samples representative of the different \"voices\" have also been enriched with meta-linguistic and syntactic information. The layer of syntactic annotation forms the first nucleus of an Italian historical treebank complying with the Universal Dependencies standard. The paper illustrates the final resource, the methodology and tools used to build it, and the Web Interface for navigating it","keywords":["Historical Corpora","Linguistic and Meta-linguistic Annotation","Information Extraction"],"pages":"911-918","url":"https:\/\/www.aclweb.org\/anthology\/2020.lrec-1.114.pdf","volume":"","doi":"","editors":[],"editors_source":"","published":"","publisher":"European Language Resources Association ELRA (Paris, FRA)","issn":"","isbn":"979-10-95546-34-4","conference_name":"Conference on Language Resources and Evaluation (LREC)","conference_place":"Paris","conference_date":"","last_updated_cnr":"2025-02-09 08:09:54","last_updated_oai":"2025-02-09 08:09:54","last_updated_www":"0000-00-00 00:00:00"},{"id":1787,"id_source":421767,"institutes":["ILC"],"type":"conference_article","type_order":7,"title":"Is Neural Language Model Perplexity Related to Readability?","year":2020,"authors":["Miaschi, A.","Alzetta, C.","Brunato, D. P.","Dell'Orletta, F.","Venturi, G."],"authors_source":"Miaschi, Alessio; Miaschi, Alessio; Alzetta, Chiara; Alzetta, Chiara; Brunato, DOMINIQUE PIERINA; Brunato, DOMINIQUE PIERINA; Dell'Orletta, Felice; Dell'Orletta, Felice; Venturi, Giulia; Venturi, Giulia","authors_cnr_name":["MIASCHI, ALESSIO","MIASCHI, ALESSIO","ALZETTA, CHIARA","ALZETTA, CHIARA","BRUNATO, DOMINIQUE PIERINA","BRUNATO, DOMINIQUE PIERINA","DELL'ORLETTA, FELICE","DELL'ORLETTA, FELICE","VENTURI, GIULIA","VENTURI, GIULIA"],"authors_cnr_id":["rp12522","rp12522","rp12530","rp12530","rp06836","rp06836","rp22811","rp22811","rp00732","rp00732"],"authors_cnr_institute":[],"abstract":"This paper explores the relationship between Neural Language Model (NLM) perplexity and sentence readability. Starting from the evidence that NLMs implicitly acquire sophisticated linguistic knowledge from a huge amount of training data, our goal is to investigate whether perplexity is affected by linguistic features used to automatically assess sentence readability and if there is a correlation between the two metrics. Our findings suggest that this correlation is actually quite weak and the two metrics are affected by different linguistic phenomena","keywords":["nlp","neural language models","readability"],"pages":"","url":"http:\/\/ceur-ws.org\/Vol-2769\/paper_57.pdf","volume":"","doi":"","editors":[],"editors_source":"","published":"Proceedings of the Seventh Italian Conference on Computational Linguistics","publisher":"","issn":"","isbn":"979-12-80136-28-2","conference_name":"Seventh Italian Conference on Computational Linguistics","conference_place":"","conference_date":"","last_updated_cnr":"2024-12-20 02:13:07","last_updated_oai":"2024-12-20 02:13:07","last_updated_www":"0000-00-00 00:00:00"},{"id":1972,"id_source":379646,"institutes":["ILC"],"type":"conference_article","type_order":7,"title":"Linguistic Profiling of a Neural Language Model","year":2020,"authors":["Miaschi, A.","Brunato, D.","Dell'Orletta, F.","Venturi, G."],"authors_source":"Miaschi, A; Brunato, D; Dell'Orletta, F; Venturi, G","authors_cnr_name":["MIASCHI, ALESSIO","BRUNATO, DOMINIQUE PIERINA","DELL'ORLETTA, FELICE","VENTURI, GIULIA"],"authors_cnr_id":["rp12522","rp06836","rp22811","rp00732"],"authors_cnr_institute":[],"abstract":"In this paper we investigate the linguistic knowledge learned by a Neural Language Model (NLM) before and after a fine-tuning process and how this knowledge affects its predictions during several classification problems. We use a wide set of probing tasks, each of which corresponds to a distinct sentence-level feature extracted from different levels of linguistic annotation. We show that BERT is able to encode a wide range of linguistic characteristics, but it tends to lose this information when trained on specific downstream tasks. We also find that BERT's capacity to encode different kind of linguistic properties has a positive influence on its predictions: the more it stores readable linguistic information of a sentence, the higher will be its capacity of predicting the expected label assigned to that sentence","keywords":["Linguistic Profiling","Neural Language Model","Interpretability"],"pages":"745-756","url":"https:\/\/www.aclweb.org\/anthology\/2020.coling-main.65\/","volume":"","doi":"10.18653\/v1\/2020.coling-main.65","editors":[],"editors_source":"","published":"International Conference on Computational Linguistics (COLING)","publisher":"","issn":"","isbn":"978-1-952148-27-9","conference_name":"International Conference on Computational Linguistics (COLING)","conference_place":"","conference_date":"","last_updated_cnr":"2025-03-13 22:58:33","last_updated_oai":"2025-03-13 22:58:33","last_updated_www":"0000-00-00 00:00:00"},{"id":1056,"id_source":384933,"institutes":["ILC"],"type":"conference_article","type_order":7,"title":"Tracking the Evolution of Written Language Competence in L2 Spanish Learners","year":2020,"authors":["Miaschi, A.","Davidson, S.","Brunato, D. P.","Dell'Orletta, F.","Sagae, K.","Sanchez Gutierrez, C. H.","Venturi, G."],"authors_source":"Miaschi, Alessio; Davidson, Sam; Brunato, DOMINIQUE PIERINA; Dell'Orletta, Felice; Sagae, Kenji; SanchezGutierrez Claudia, H; Venturi, Giulia","authors_cnr_name":["MIASCHI, ALESSIO","BRUNATO, DOMINIQUE PIERINA","DELL'ORLETTA, FELICE","VENTURI, GIULIA"],"authors_cnr_id":["rp12522","rp06836","rp22811","rp00732"],"authors_cnr_institute":[],"abstract":"In this paper we present an NLP-based approach for tracking the evolution of written language competence in L2 Spanish learners using a wide range of linguistic features automatically extracted from students' written productions. Beyond reporting classification results for different scenarios, we explore the connection between the most predictive features and the teaching curriculum, finding that our set of linguistic features often reflects the explicit instruction that students receive during each course","keywords":["Evolution of Language Competence","Natural Language Processing","Linguistic Profiling"],"pages":"92-101","url":"https:\/\/www.aclweb.org\/anthology\/2020.bea-1.9.pdf","volume":"","doi":"10.18653\/v1\/W16-05","editors":[],"editors_source":"","published":"Proceedings of 15th Workshop on Innovative Use of NLP for Building Educational Applications","publisher":"Association for Computational Linguistics (Stroudsburg, USA)","issn":"","isbn":"978-1-941643-83-9","conference_name":"15th Workshop on Innovative Use of NLP for Building Educational Applications","conference_place":"Stroudsburg","conference_date":"","last_updated_cnr":"2025-03-19 22:34:12","last_updated_oai":"2025-03-19 22:34:12","last_updated_www":"0000-00-00 00:00:00"},{"id":1129,"id_source":421765,"institutes":["ILC"],"type":"conference_article","type_order":7,"title":"Italian Transformers Under the Linguistic Lens","year":2020,"authors":["Miaschi, A.","Sarti, G.","Brunato, D. P.","Dell'Orletta, F.","Venturi, G."],"authors_source":"Miaschi, Alessio; Miaschi, Alessio; Sarti, ; Gabriele, ; Brunato, DOMINIQUE PIERINA; Brunato, DOMINIQUE PIERINA; Dell'Orletta, Felice; Dell'Orletta, Felice; Venturi, Giulia; Venturi, Giulia","authors_cnr_name":["MIASCHI, ALESSIO","MIASCHI, ALESSIO","BRUNATO, DOMINIQUE PIERINA","BRUNATO, DOMINIQUE PIERINA","DELL'ORLETTA, FELICE","DELL'ORLETTA, FELICE","VENTURI, GIULIA","VENTURI, GIULIA"],"authors_cnr_id":["rp12522","rp12522","rp06836","rp06836","rp22811","rp22811","rp00732","rp00732"],"authors_cnr_institute":[],"abstract":"In this paper we present an in-depth investigation of the linguistic knowledge encoded by the transformer models currently available for the Italian language. In particular, we investigate whether and how using different architectures of probing models affects the performance of Italian transformers in encoding a wide spectrum of linguistic features. Moreover, we explore how this implicit knowledge varies according to different textual genres","keywords":["nlp","neural language models","interpretability"],"pages":"","url":"http:\/\/ceur-ws.org\/Vol-2769\/paper_56.pdf","volume":"","doi":"","editors":[],"editors_source":"","published":"Proceedings of the Seventh Italian Conference on Computational Linguistics (CLiC-it)","publisher":"","issn":"","isbn":"979-12-80136-28-2","conference_name":"Seventh Italian Conference on Computational Linguistics (CLiC-it)","conference_place":"","conference_date":"","last_updated_cnr":"2024-12-20 02:17:38","last_updated_oai":"2024-12-20 02:17:38","last_updated_www":"0000-00-00 00:00:00"},{"id":2128,"id_source":403586,"institutes":["ILC"],"type":"journal_article","type_order":1,"title":"INFERRING QUANTITATIVE TYPOLOGICAL TRENDS FROM MULTILINGUAL TREEBANKS. A CASE STUDY","year":2019,"authors":["Alzetta, C.","Dell'Orletta, F.","Montemagni, S.","Venturi, G."],"authors_source":"Alzetta, Chiara; Dell'Orletta, Felice; Montemagni, Simonetta; Venturi, Giulia","authors_cnr_name":["ALZETTA, CHIARA","DELL'ORLETTA, FELICE","MONTEMAGNI, SIMONETTA","VENTURI, GIULIA"],"authors_cnr_id":["rp12530","rp22811","rp16780","rp00732"],"authors_cnr_institute":[],"abstract":"In the past decades, linguistic typology went through a renewing phase that involved a significant change in the research questions and methods of the discipline, which is now interested in fine-grained features underlying language diversity. In this paper, we propose a novel approach to address the newly defined needs of linguistic typology by extracting qualitative and quantitative information about a wide range of features from multilingual annotated corpora based on Natural Language Processing methods and techniques. We tested our method in a case study focusing on word order variation in two widely investigated constructions, VERB-SUBJ(ect) and NOUN-ADJ(ective), with a specific view to structural and functional factors underlying the preference for one or the other order, both intra-and cross-linguistically, and their interaction. Preliminary experiments have been carried out aimed at acquiring typological evidence from a selection of linguistically annotated treebanks for three different languages, namely Italian, Spanish and English. Our results show the effectiveness of the method in letting similarities and differences also emerge from typologically close languages","keywords":["language typology","multilingual annotated corpora","linguistic knowledge extraction and modelling","word order variation"],"pages":"209-242","url":"https:\/\/www.rivisteweb.it\/doi\/10.1418\/95391","volume":"18 (2)","doi":"10.1418\/95391","editors":[],"editors_source":"","published":"LINGUE E LINGUAGGIO","publisher":"","issn":"1720-9331","isbn":"","conference_name":"","conference_place":"","conference_date":"","last_updated_cnr":"2025-06-15 00:46:01","last_updated_oai":"2025-06-15 00:46:01","last_updated_www":"0000-00-00 00:00:00"},{"id":78,"id_source":403580,"institutes":["ILC"],"type":"journal_article","type_order":1,"title":"Le parole del miglioramento. Come le scuole descrivono il cambiamento","year":2019,"authors":["Dell'Orletta, F.","Greco, S.","Montemagni, S.","Morini, E.","Rossi, F.","Sagri, M.","Venturi, G."],"authors_source":"Dell'Orletta, F; Greco, S; Montemagni, S; Morini, E; Rossi, F; Sagri, Mt; Venturi, G","authors_cnr_name":["DELL'ORLETTA, FELICE","MONTEMAGNI, SIMONETTA","VENTURI, GIULIA"],"authors_cnr_id":["rp22811","rp16780","rp00732"],"authors_cnr_institute":[],"abstract":"Il presente contributo intende illustrare i risultati di una ricerca condotta con l'uso di strumenti di trattamento automatico del linguaggio (Natural Language Processing: nlp) su quanto dichiarato dalle scuole in circa 2500 Piani di Miglioramento (modello indire) con l'obiettivo di comprendere le scelte strategiche in un'ottica di miglioramento continuo. Il disegno d'analisi permette di restituire sia una visione complessiva dei Piani di Miglioramento che approfondimenti qualitativi di confronto tra tipologie di scuola e aree geografiche e relativi a tematiche strategiche quali formazione e innovazione","keywords":["Piano d","Natural Language Processing","Formazione","Innovazione"],"pages":"47-68","url":"https:\/\/www.rivistainfanzia.it\/pvw\/app\/default\/pvw_sito.php?sede_codice=1PWPSE01&page=2432193","volume":"1\/2019","doi":"","editors":[],"editors_source":"","published":"PSICOLOGIA DELL'EDUCAZIONE","publisher":"","issn":"1971-3711","isbn":"","conference_name":"","conference_place":"","conference_date":"","last_updated_cnr":"2024-04-05 10:55:50","last_updated_oai":"2024-04-05 10:55:50","last_updated_www":"0000-00-00 00:00:00"},{"id":1998,"id_source":403584,"institutes":["ILC"],"type":"journal_article","type_order":1,"title":"Analytics dei testi riflessivi scritti dai docenti neoassunti nel portfolio digitale","year":2019,"authors":["Della Gala, V.","Chiriatti, G.","Dell'Orletta, F.","Pettenati, M. C.","Venturi, G."],"authors_source":"Della Gala V.; Chiriatti G.; Dell'Orletta F.; Pettenati M.C.; Venturi G.","authors_cnr_name":["DELL'ORLETTA, FELICE","VENTURI, GIULIA"],"authors_cnr_id":["rp22811","rp00732"],"authors_cnr_institute":[],"abstract":"Presentiamo i risultati preliminari e l'analisi svolta su circa 50. 000 testi scritti dai docenti neo nominati in ruolo per riflettere su due attivit\u00e0 didattiche svolte con gli studenti, nel contesto del percorso dell'anno di formazione e prova 2016\/17. Il percorso prevede attivit\u00e0 in presenza e attivit\u00e0 a distanza completate sul portfolio digitale, ospitato nell'ambiente online gestito dall'Indire. Nell'ambito del monitoraggio della formazione, con il fine di ottimizzare gli strumenti e il supporto fornito, abbiamo interrogato i dati testuali prodotti dai docenti nell'interazione con l'ambiente per capire se i testi presentassero evidenze riconducibili alle scritture riflessive. Obiettivi dell'indagine sono stati la definizione di uno schema per la classificazione dei testi sulla base del livello di riflessivit\u00e0 evidenziato e l'impiego di strumenti di Trattamento Automatico del Linguaggio (TAL) per l'analisi dell'interocorpus testuale prodotto dai docenti. Descriveremo il contesto scientifico e progettuale, le caratteristiche dei dati analizzati, come questo abbia determinato il disegno d'indagine; descriveremo inoltre la sua implementazione e dunque le procedure, gli strumenti e le metriche adottate o elaborate per rappresentare il contenuto dei dati; infine discuteremo i primi risultati e alcuni vantaggi e limiti dell'approccio adottato","keywords":["Teacher professional development","Natural Language Processing","Reflective writing","Linguistic Profiling","Document Classification"],"pages":"187-204","url":"https:\/\/ojs.pensamultimedia.it\/index.php\/sird\/article\/view\/3454\/3360","volume":"SPECIAL ISSUE","doi":"10.7346\/SIRD-2S2019-P189","editors":[],"editors_source":"","published":"GIORNALE ITALIANO DELLA RICERCA EDUCATIVA (ONLINE)","publisher":"","issn":"2038-9744","isbn":"","conference_name":"","conference_place":"","conference_date":"","last_updated_cnr":"2024-06-22 13:32:11","last_updated_oai":"2024-06-22 13:32:11","last_updated_www":"0000-00-00 00:00:00"},{"id":331,"id_source":403587,"institutes":["ILC"],"type":"conference_article","type_order":7,"title":"Dissecting Treebanks to Uncover Typological Trends. A Multilingual Comparative Approach","year":2019,"authors":["Alzetta, C.","Dell'Orletta, F.","Montemagni, S.","Venturi, G."],"authors_source":"Alzetta C.; Dell'Orletta F.; Montemagni S.; Venturi G.","authors_cnr_name":["ALZETTA, CHIARA","DELL'ORLETTA, FELICE","MONTEMAGNI, SIMONETTA","VENTURI, GIULIA"],"authors_cnr_id":["rp12530","rp22811","rp16780","rp00732"],"authors_cnr_institute":[],"abstract":"Over the last years, linguistic typology started attracting the interest of the community working on cross-and multi-lingual NLP as a way to tackle the bottleneck deriving from the lack of annotated data for many languages. Typological information is mostly acquired from publicly accessible typological databases, manually constructed by linguists. As reported in Ponti et al. (2018), despite the abundant information contained in them for many languages, these resources suffer from two main shortcomings, i. e. their limited coverage and the discrete nature of features (only \"the majority value rather than the full range of possible values and their corresponding frequencies\" is reported). Corpus-based studies can help to automatically acquire quantitative typological evidence which might be exploited for polyglot NLP. Recently, the availability of corpora annotated following a cross-linguistically consistent annotation scheme such as the one developed in the Universal Dependencies project is prompting new comparative linguistic studies aimed to identify similarities as well as idiosyncrasies among typologically different languages (Nivre, 2015). The line of research described here is aimed at acquiring quantitative typological evidence from UD treebanks through a multilingual contrastive approach","keywords":["Natural Language Processing","Linguistic Typology"],"pages":"1-3","url":"https:\/\/typology-and-nlp.github.io\/2019\/assets\/2019\/papers\/5.pdf","volume":"","doi":"","editors":[],"editors_source":"","published":"","publisher":"","issn":"","isbn":"978-1-950737-29-1","conference_name":"1st TyP-NLP: The Workshop on Typology for Polyglot NLP, ACL workshop","conference_place":"","conference_date":"","last_updated_cnr":"2024-10-27 08:04:42","last_updated_oai":"2024-10-27 08:04:42","last_updated_www":"0000-00-00 00:00:00"},{"id":974,"id_source":392228,"institutes":["ILC"],"type":"conference_article","type_order":7,"title":"Nove Anni di jTEI: What's New?","year":2019,"authors":["Boschetti, F.","Pardelli, G.","Venturi, G."],"authors_source":"Boschetti, Federico; Pardelli, Gabriella; Venturi, Giulia","authors_cnr_name":["BOSCHETTI, FEDERICO","PARDELLI, GABRIELLA","VENTURI, GIULIA"],"authors_cnr_id":["rp04876","rp20820","rp00732"],"authors_cnr_institute":[],"abstract":"This paper illustrates methods and tools to study the development of research topics in the TEI community across the years. For this purpose, automatic terminology extraction technologies were exploited","keywords":["Natural Language Processing","Digital Humanities"],"pages":"1-6","url":"http:\/\/ceur-ws.org\/Vol-2481","volume":"","doi":"","editors":["Bernardi, R.","Navigli, R.","Semeraro, G."],"editors_source":"Raffaella Bernardi, Roberto Navigli, Giovanni Semeraro","published":"CLiC-it 2019 Italian Conference on Computational Linguistics","publisher":"CEUR-WS. org (Aachen, DEU)","issn":"","isbn":"","conference_name":"CLiC-it 2019-Sesta Conferenza Italiana di Linguistica Computazionale","conference_place":"Aachen","conference_date":"","last_updated_cnr":"2024-06-11 13:26:10","last_updated_oai":"2024-06-11 13:26:10","last_updated_www":"0000-00-00 00:00:00"},{"id":1391,"id_source":380338,"institutes":["ILC"],"type":"conference_article","type_order":7,"title":"What makes a review helpful? Predicting the helpfulness of Italian tripadvisor reviews","year":2019,"authors":["Chiriatti, G.","Brunato, D.","Dell'Orletta, F.","Venturi, G."],"authors_source":"Chiriatti G.; Brunato D.; Dell'Orletta F.; Venturi G.","authors_cnr_name":["BRUNATO, DOMINIQUE PIERINA","DELL'ORLETTA, FELICE","VENTURI, GIULIA"],"authors_cnr_id":["rp06836","rp22811","rp00732"],"authors_cnr_institute":[],"abstract":"In this paper we introduce a classification system devoted to predict the helpfulness of Italian online reviews. It is based on a wide set of features reflecting the different factors involved and tested on different categories of TripAdvisor reviews. For this purpose, we collected the first Italian corpus of online reviews enriched with metadata related to their helpfulness and we carried out an in-depth analysis of the most predictive features","keywords":["Natural Language Processing","Documenti Classification","Linguistic Profiling"],"pages":"1-6","url":"http:\/\/www.scopus.com\/record\/display.url?eid=2-s2.0-85074834351&origin=inward","volume":"2481","doi":"","editors":[],"editors_source":"","published":"","publisher":"","issn":"","isbn":"","conference_name":"6th Italian Conference on Computational Linguistics (CLiC-it)","conference_place":"","conference_date":"","last_updated_cnr":"2024-06-02 12:13:11","last_updated_oai":"2024-06-02 12:13:11","last_updated_www":"0000-00-00 00:00:00"},{"id":1347,"id_source":380336,"institutes":["ILC"],"type":"conference_article","type_order":7,"title":"Italian and English sentence simplification: How many differences?","year":2019,"authors":["Fieromonte, M.","Brunato, D.","Dell'Orletta, F.","Venturi, G."],"authors_source":"Fieromonte M.; Brunato D.; Dell'Orletta F.; Venturi G.","authors_cnr_name":["BRUNATO, DOMINIQUE PIERINA","DELL'ORLETTA, FELICE","VENTURI, GIULIA"],"authors_cnr_id":["rp06836","rp22811","rp00732"],"authors_cnr_institute":[],"abstract":"The paper proposes a cross-linguistic analysis of two parallel monolingual corpora conceived for automatic text simplification in two languages, Italian and English. The aim is to find similarities and differences in the process of simplification in two typologically different languages. To carry out the comparison, 1, 000 sentences were extracted from the two corpora and annotated with a scheme previously used to annotate simplification phenomena","keywords":["Natural Language Processing","Automatic Text Simplification"],"pages":"1-6","url":"http:\/\/www.scopus.com\/record\/display.url?eid=2-s2.0-85074816689&origin=inward","volume":"2481","doi":"","editors":[],"editors_source":"","published":"","publisher":"","issn":"","isbn":"","conference_name":"6th Italian Conference on Computational Linguistics (CLiC-it)","conference_place":"","conference_date":"","last_updated_cnr":"2024-03-26 19:45:34","last_updated_oai":"2024-03-26 19:45:34","last_updated_www":"0000-00-00 00:00:00"},{"id":665,"id_source":403573,"institutes":["ILC","IGSG"],"type":"book_chapter","type_order":4,"title":"Semantic processing of legal texts","year":2018,"authors":["Agnoloni, T.","Venturi, G."],"authors_source":"Agnoloni, T; Venturi, G","authors_cnr_name":["AGNOLONI, TOMMASO","VENTURI, GIULIA"],"authors_cnr_id":["rp21506","rp00732"],"authors_cnr_institute":[],"abstract":"The paper provides an overview of the field of semantic processing of legal texts, combining views and perspectives from the computational linguistic and Artificial Intelligence and Law (AI & Law) communities. The last few years have seen a growing body of research and practice in the field of AI & Law which addresses a range of topics: semantic and cross-language legal Information Retrieval, document classification, legal drafting, legal knowledge extraction, automated legal argumentation, as well as the construction of legal ontologies and their application. The increasing availability of legal corpora accessible as processable data is making viable their partially automated conversion into legal knowledge bases. In this context, it is of paramount importance the use of Natural Language Processing (NLP) techniques and tools that automate the process of knowledge extraction from legal texts. Accordingly, the paper aims at discussing how the two research communities can benefit from the interaction of the different perspectives: the legal artificial intelligence community can gain insight into state-of-the-art linguistic technologies, tools and resources, and the computational linguists can take advantage of the large and often multilingual legal resources (corpora as well as lexicons and ontologies) for training, domain adaptation and evaluation of current NLP technologies and tools. The authors will present an overview on semantic resources for legal texts annotation and processing. Different kind of resources (linguistic, lexical, conceptual, formal) will be introduced and their differences, methodological premises, intended use and possible integration will be highlighted. The peculiarities of the legal domain and legal language will be discussed in relation with the construction and use of legal semantic resources. The issue of multilingualism, multilingual and multi-legal system access to legal information will be also discussed showing how formalized lexical, linguistic and conceptual legal resources can support the task. How NLP tools and techniques can be fruitfully exploited to semantically process collections of legal texts will be introduced in the second part of the paper. In particular, the authors will show how they can be used to automatically extract the relevant knowledge contained in legal text corpora, to structure the extracted knowledge in semantic resources (such as domain-specific ontologies or thesauri), and to semantically annotate the texts with the extracted information to pave the way to content-based access and querying","keywords":["Semantic Processing","Natural Language Processing","Ontology Learning","Legal Texts"],"pages":"109-137","url":"http:\/\/www.scopus.com\/record\/display.url?eid=2-s2.0-85061292435&origin=inward","volume":"","doi":"10.1515\/9781614514664-006","editors":[],"editors_source":"","published":"","publisher":"Walter De Gruyter Inc (Boston\/Berlin\/Munich, USA)","issn":"","isbn":"978-1-61451-669-9","conference_name":"","conference_place":"Boston\/Berlin\/Munich","conference_date":"","last_updated_cnr":"2024-06-11 12:49:46","last_updated_oai":"2024-06-11 12:49:46","last_updated_www":"0000-00-00 00:00:00"},{"id":1816,"id_source":374898,"institutes":["ILC"],"type":"conference_article","type_order":7,"title":"Extracting dependency relations from digital learning content","year":2018,"authors":["Adorni, G.","Dell'Orletta, F.","Koceva, F.","Torre, I.","Venturi, G."],"authors_source":"Adorni G.; Dell'Orletta F.; Koceva F.; Torre I.; Venturi G.","authors_cnr_name":["DELL'ORLETTA, FELICE","VENTURI, GIULIA"],"authors_cnr_id":["rp22811","rp00732"],"authors_cnr_institute":[],"abstract":"Digital Libraries present tremendous potential for developing e-learning applications, such as text comprehension and question-answering tools. A way to build this kind of tools is structuring the digital content into relevant concepts and dependency relations among them. While the literature offers several approaches for the former, the identification of dependencies, and specifically of prerequisite relations, is still an open issue. We present an approach to manage this task","keywords":["Prerequisite relationship","Concept extraction","Graph mining"],"pages":"114-119","url":"http:\/\/www.scopus.com\/record\/display.url?eid=2-s2.0-85041860435&origin=inward","volume":"806","doi":"10.1007\/978-3-319-73165-0_11","editors":[],"editors_source":"","published":"","publisher":"","issn":"","isbn":"","conference_name":"14th Italian Research Conference on Digital Libraries (IRCDL 2018)","conference_place":"","conference_date":"","last_updated_cnr":"2024-05-19 13:59:59","last_updated_oai":"2024-05-19 13:59:59","last_updated_www":"0000-00-00 00:00:00"},{"id":271,"id_source":371344,"institutes":["ILC"],"type":"conference_article","type_order":7,"title":"Assessing the Impact of Iterative Error Detection and Correction. A Case Study on the Italian Universal Dependency Treebank","year":2018,"authors":["Alzetta, C.","Dell'Orletta, F.","Montemagni, S.","Simi, M.","Venturi, G."],"authors_source":"Alzetta, C; Dell'Orletta, F; Montemagni, S; Simi, M; Venturi, G","authors_cnr_name":["ALZETTA, CHIARA","DELL'ORLETTA, FELICE","MONTEMAGNI, SIMONETTA","VENTURI, GIULIA"],"authors_cnr_id":["rp12530","rp22811","rp16780","rp00732"],"authors_cnr_institute":[],"abstract":"Detection and correction of errors and inconsistencies in \"gold treebanks\" are becoming more and more central topics of corpus annotation. The paper illustrates a new incremental method for enhancing treebanks, with particular emphasis on the extension of error patterns across different textual genres and registers. Impact and role of corrections have been assessed in a dependency parsing experiment carried out with four different parsers, whose results are promising. For both evaluation datasets, the performance of parsers increases, in terms of the standard LAS and UAS measures and of a more focused measure taking into account only relations involved in error patterns, and at the level of individual dependencies","keywords":["Error Detection","Universal Dependency Treebanks","Syntactic parsing"],"pages":"1-7","url":"http:\/\/universaldependencies.org\/udw18\/PDFs\/39_Paper.pdf","volume":"","doi":"","editors":[],"editors_source":"","published":"","publisher":"","issn":"","isbn":"978-1-948087-84-1","conference_name":"Universal Dependencies Workshop 2018 (UDW 2018)","conference_place":"","conference_date":"","last_updated_cnr":"2024-07-21 07:03:38","last_updated_oai":"2024-07-21 07:03:38","last_updated_www":"0000-00-00 00:00:00"},{"id":1372,"id_source":493647,"institutes":["ILC","ISTI"],"type":"conference_article","type_order":7,"title":"Assessing the Impact of Incremental Error Detection and Correction. A Case Study on the Italian Universal Dependency Treebank","year":2018,"authors":["Alzetta, C.","Dell'Orletta, F.","Montemagni, S.","Simi, M.","Venturi, G."],"authors_source":"Alzetta, C.; Dell'Orletta, F.; Montemagni, S.; Simi, M.; Venturi, G.","authors_cnr_name":["ALZETTA, CHIARA","DELL'ORLETTA, FELICE","MONTEMAGNI, SIMONETTA","VENTURI, GIULIA"],"authors_cnr_id":["rp12530","rp22811","rp16780","rp00732"],"authors_cnr_institute":[],"abstract":"Detection and correction of errors and inconsistencies in \u201cgold treebanks\u201d are becoming more and more central topics of corpus annotation. The paper illustrates a new incremental method for enhancing treebanks, with particular emphasis on the extension of error patterns across different textual genres and registers. Impact and role of corrections have been assessed in a dependency parsing experiment carried out with four different parsers, whose results are promising. For both evaluation datasets, the performance of parsers increases, in terms of the standard LAS and UAS measures and of a more focused measure taking into account only relations involved in error patterns, and at the level of individual dependencies","keywords":["Treebank, annotation, annotation error"],"pages":"1-7","url":"https:\/\/iris.cnr.it\/handle\/20.500.14243\/493647","volume":"","doi":"","editors":[],"editors_source":"","published":"EMNLP 2018-2nd Workshop on Universal Dependencies, UDW 2018-Proceedings of the Workshop","publisher":"Association for Computational Linguistics (ACL)","issn":"","isbn":"9781948087780","conference_name":"2nd Workshop on Universal Dependencies, UDW 2018, held in conjunction with EMNLP 2018","conference_place":"","conference_date":"","last_updated_cnr":"2024-12-04 00:30:14","last_updated_oai":"2024-12-04 00:30:14","last_updated_www":"0000-00-00 00:00:00"},{"id":1332,"id_source":334766,"institutes":["ILC"],"type":"conference_article","type_order":7,"title":"Dangerous Relations in Dependency Treebanks","year":2018,"authors":["Alzetta, C.","Dell'Orletta, F.","Montemagni, S.","Venturi, G."],"authors_source":"Alzetta, Chiara; Dell'Orletta, Felice; Montemagni, Simonetta; Venturi, Giulia","authors_cnr_name":["ALZETTA, CHIARA","DELL'ORLETTA, FELICE","MONTEMAGNI, SIMONETTA","VENTURI, GIULIA"],"authors_cnr_id":["rp12530","rp22811","rp16780","rp00732"],"authors_cnr_institute":[],"abstract":"The paper illustrates an effective and innovative method for detecting erroneously annotated arcs in gold dependency treebanks based on an algorithm originally developed to measure the reliability of automatically produced dependency relations. The method permits to significantly restrict the error search space and, more importantly, to reliably identify patterns of systematic recurrent errors which represent dangerous evidence to a parser which tendentially will replicate them. Achieved results demonstrate effectiveness and reliability of the method","keywords":["Dependency treebanks","Error Detection","Linguistic Annotation"],"pages":"201-210","url":"http:\/\/aclweb.org\/anthology\/W\/W17\/W17-7624.pdf","volume":"","doi":"","editors":[],"editors_source":"","published":"Proceedings of the 16th International Workshop on Treebanks and Linguistic Theories","publisher":"","issn":"","isbn":"978-80-88132-04-2","conference_name":"16th International Workshop on Treebanks and Linguistic Theories","conference_place":"","conference_date":"","last_updated_cnr":"2024-07-23 11:21:46","last_updated_oai":"2024-07-23 11:21:46","last_updated_www":"0000-00-00 00:00:00"},{"id":1790,"id_source":374901,"institutes":["ILC"],"type":"conference_article","type_order":7,"title":"Universal Dependencies and Quantitative Typological Trends. A Case Study on Word Order","year":2018,"authors":["Alzetta, C.","Dell'Orletta, F.","Montemagni, S.","Venturi, G."],"authors_source":"Alzetta, Chiara; Dell'Orletta, Felice; Montemagni, Simonetta; Venturi, Giulia","authors_cnr_name":["ALZETTA, CHIARA","DELL'ORLETTA, FELICE","MONTEMAGNI, SIMONETTA","VENTURI, GIULIA"],"authors_cnr_id":["rp12530","rp22811","rp16780","rp00732"],"authors_cnr_institute":[],"abstract":"The paper presents a new methodology aimed at acquiring typological evidence from \"gold\" treebanks for different languages. In particular, it investigates whether and to what extent algorithms developed for assessing the plausibility of automatically produced syntactic annotations could contribute to shed light on key issues of the linguistic typological literature. It reports the first and promising results of a case study focusing on word order patterns carried out on three different languages (English, Italian and Spanish)","keywords":["Linguistic Knowledge Extraction","Dependency Treebanks","Linguistic Typology"],"pages":"4540-4549","url":"http:\/\/www.lrec-conf.org\/proceedings\/lrec2018\/pdf\/1109.pdf","volume":"","doi":"","editors":[],"editors_source":"","published":"","publisher":"European Language Resources Association ELRA (Paris, FRA)","issn":"","isbn":"979-10-95546-00-9","conference_name":"Proceedings of the 11th Edition of the Language Resources and Evaluation Conference (LREC 2018)","conference_place":"Paris","conference_date":"","last_updated_cnr":"2024-07-09 11:30:29","last_updated_oai":"2024-07-09 11:30:29","last_updated_www":"0000-00-00 00:00:00"},{"id":1830,"id_source":371346,"institutes":["ILC"],"type":"conference_article","type_order":7,"title":"Is this sentence difficult? Do you agree?","year":2018,"authors":["Brunato, D.","De Mattei, L.","Dell'Orletta, F.","Iavarone, B.","Venturi, G."],"authors_source":"Brunato D.; De Mattei L.; Dell'Orletta F.; Iavarone B.; Venturi G.","authors_cnr_name":["BRUNATO, DOMINIQUE PIERINA","DELL'ORLETTA, FELICE","VENTURI, GIULIA"],"authors_cnr_id":["rp06836","rp22811","rp00732"],"authors_cnr_institute":[],"abstract":"In this paper, we present a crowdsourcing-based approach to model the human perception of sentence complexity. We collect a large corpus of sentences rated with judgments of complexity for two typologically-different languages, Italian and English. We test our approach in two experimental scenarios aimed to investigate the contribution of a wide set of lexical, morpho-syntactic and syntactic phenomena in predicting i) the degree of agreement among annotators independently from the assigned judgment and ii) the perception of sentence complexity","keywords":["Linguistic complexity","Crowdsourcing","Human perception"],"pages":"1-10","url":"https:\/\/www.aclweb.org\/anthology\/D18-1289\/","volume":"","doi":"10.18653\/v1\/D18-1289","editors":[],"editors_source":"","published":"","publisher":"Association for Computational Linguistics (Stroudsburg, USA)","issn":"","isbn":"978-1-948087-84-1","conference_name":"Conference on Empirical Methods in Natural Language Processing (EMNLP)","conference_place":"Stroudsburg","conference_date":"","last_updated_cnr":"2024-06-16 12:37:32","last_updated_oai":"2024-06-16 12:37:32","last_updated_www":"0000-00-00 00:00:00"},{"id":290,"id_source":403577,"institutes":["ILC"],"type":"conference_article","type_order":7,"title":"A NLP-based analysis of reflective writings by Italian teachers","year":2018,"authors":["Chiriatti, G.","Della Gala, V.","Dell'Orletta, F.","Montemagni, S.","Pettenati, M. C.","Sagri, M. T.","Venturi, G."],"authors_source":"Chiriatti G.; Della Gala V.; Dell'Orletta F.; Montemagni S.; Pettenati M.C.; Sagri M.T.; Venturi G.","authors_cnr_name":["DELL'ORLETTA, FELICE","MONTEMAGNI, SIMONETTA","VENTURI, GIULIA"],"authors_cnr_id":["rp22811","rp16780","rp00732"],"authors_cnr_institute":[],"abstract":"This paper reports first results of a wider study devoted to exploit the potentialities of a NLP-based approach to the analysis of a corpus of reflective writings on teaching activities. We investigate how a wide set of linguistic features allows reconstructing the linguistic profile of the texts written by the Italian teachers and predicting whether are reflective","keywords":["Natural Language Processing","Reflective Writings","Linguistic Profiling","Document Classification"],"pages":"1-7","url":"http:\/\/www.scopus.com\/record\/display.url?eid=2-s2.0-85057733802&origin=inward","volume":"2253","doi":"","editors":[],"editors_source":"","published":"","publisher":"","issn":"","isbn":"","conference_name":"5th Italian Conference on Computational Linguistics (CLiC-it)","conference_place":"","conference_date":"","last_updated_cnr":"2024-06-18 14:58:54","last_updated_oai":"2024-06-18 14:58:54","last_updated_www":"0000-00-00 00:00:00"},{"id":2079,"id_source":403576,"institutes":["ILC"],"type":"conference_article","type_order":7,"title":"Sentences and documents in native language identification","year":2018,"authors":["Cimino, A.","Dell'Orletta, F.","Brunato, D.","Venturi, G."],"authors_source":"Cimino A.; Dell'Orletta F.; Brunato D.; Venturi G.","authors_cnr_name":["CIMINO, ANDREA","DELL'ORLETTA, FELICE","BRUNATO, DOMINIQUE PIERINA","VENTURI, GIULIA"],"authors_cnr_id":["rp05770","rp22811","rp06836","rp00732"],"authors_cnr_institute":[],"abstract":"Starting from a wide set of linguistic features, we present the first in depth feature analysis in two different Native Language Identification (NLI) scenarios. We compare the results obtained in a traditional NLI document classification task and in a newly introduced sentence classification task, investigating the different role played by the considered features. Finally, we study the impact of a set of selected features extracted from the sentence classifier in document classification","keywords":["Natural Language Processing","Native Language Identification"],"pages":"1-6","url":"http:\/\/www.scopus.com\/record\/display.url?eid=2-s2.0-85057749754&origin=inward","volume":"2253","doi":"","editors":[],"editors_source":"","published":"","publisher":"","issn":"","isbn":"","conference_name":"5th Italian Conference on Computational Linguistics (CLiC-it)","conference_place":"","conference_date":"","last_updated_cnr":"2024-05-18 12:57:39","last_updated_oai":"2024-05-18 12:57:39","last_updated_www":"0000-00-00 00:00:00"},{"id":344,"id_source":403579,"institutes":["ILC"],"type":"conference_article","type_order":7,"title":"Gender and Genre Linguistic profiling: A case study on female and male journalistic and diary prose","year":2018,"authors":["Cocciu, E.","Brunato, D.","Venturi, G.","Dell'Orletta, F."],"authors_source":"Cocciu, E; Brunato, D; Venturi, G; Dell'Orletta, F","authors_cnr_name":["BRUNATO, DOMINIQUE PIERINA","VENTURI, GIULIA","DELL'ORLETTA, FELICE"],"authors_cnr_id":["rp06836","rp00732","rp22811"],"authors_cnr_institute":[],"abstract":"This paper intends to investigate the linguistic profile of male-and female-authored texts belonging to two very different textual genres: newspaper articles and diary prose. By using a wide set of linguistic features automatically extracted from text and spanning across different levels of linguistic description, from lexicon to syntax, our analysis highlights the peculiarities of the two examined genres and how the genre dimension is influenced by variation depending on author's gender (and vice versa)","keywords":["Natural Language Processing","Genre Classification","Linguistic Profiling"],"pages":"1-6","url":"http:\/\/www.scopus.com\/record\/display.url?eid=2-s2.0-85057759773&origin=inward","volume":"2253","doi":"","editors":[],"editors_source":"","published":"","publisher":"","issn":"","isbn":"","conference_name":"5th Italian Conference on Computational Linguistics (CLiC-it)","conference_place":"","conference_date":"","last_updated_cnr":"2024-05-17 15:20:07","last_updated_oai":"2024-05-17 15:20:07","last_updated_www":"0000-00-00 00:00:00"},{"id":1889,"id_source":403578,"institutes":["ILC"],"type":"conference_article","type_order":7,"title":"Italian in the Trenches: Linguistic annotation and analysis of texts of the great war","year":2018,"authors":["De Felice, I.","Dell'Orletta, F.","Venturi, G.","Lenci, A.","Montemagni, S."],"authors_source":"De Felice I.; Dell'Orletta F.; Venturi G.; Lenci A.; Montemagni S.","authors_cnr_name":["DELL'ORLETTA, FELICE","VENTURI, GIULIA","MONTEMAGNI, SIMONETTA"],"authors_cnr_id":["rp22811","rp00732","rp16780"],"authors_cnr_institute":[],"abstract":"The paper illustrates the design and development of a textual corpus representative of the historical variants of Italian during the Great War, which was enriched with linguistic (lemmatization and pos-tagging) and meta-linguistic annotation. The corpus, after a manual revision of the linguistic annotation, was used for specializing existing NLP tools to process historical texts with promising results","keywords":["Natural Language Processing","Automatic Linguistic Annotation"],"pages":"1-5","url":"http:\/\/www.scopus.com\/record\/display.url?eid=2-s2.0-85057734451&origin=inward","volume":"2253","doi":"","editors":[],"editors_source":"","published":"","publisher":"","issn":"","isbn":"","conference_name":"5th Italian Conference on Computational Linguistics (CLiC-it)","conference_place":"","conference_date":"","last_updated_cnr":"2024-05-17 17:53:56","last_updated_oai":"2024-05-17 17:53:56","last_updated_www":"0000-00-00 00:00:00"},{"id":416,"id_source":342151,"institutes":["ILC"],"type":"journal_article","type_order":1,"title":"La qualit\u00e0 dei consensi informati. Un'analisi linguistico-computazionale della leggibilit\u00e0 dei testi","year":2017,"authors":["Venturi, G.","Dell'Orletta, F.","Montemagni, S.","Flore, E.","Bellandi, T."],"authors_source":"Venturi, G; Dell'Orletta, F; Montemagni, S; Flore, E; Bellandi, T","authors_cnr_name":["VENTURI, GIULIA","DELL'ORLETTA, FELICE","MONTEMAGNI, SIMONETTA"],"authors_cnr_id":["rp00732","rp22811","rp16780"],"authors_cnr_institute":[],"abstract":"La leggibilit\u00e0 dei testi delle informative di consenso per le procedure diagnostico-terapeutiche \u00e8 un requisito fondamentale, per offrire alle persone assistite l'accesso alle informazioni necessarie a una scelta consapevole delle opzioni disponibili per curare i diversi problemi di salute. La disponibilit\u00e0 di un testo leggibile \u00e8 inoltre un aiuto per i medici responsabili della comunicazione e della raccolta del consenso, che possono impiegarlo come un ausilio alle informazioni presentate in forma verbale durante il colloquio, in modo tale da poter condividere una base di conoscenze minime da condividere con il paziente e i suoi familiari. Seppure le evidenze siano limitate in merito alla relazione tra la qualit\u00e0 del consenso e l'attitudine al contenzioso da parte dei pazienti in caso di trattamenti che esitano in un danno attribuibile alle cure (Durand et al., 2015), si tratta di un ambito di ricerca di crescente interesse nella letteratura sulla sicurezza (Wu et al., 2005; Manta et al., 2017). Nella casistica regionale della Toscana sulle richieste di risarcimento, solo l'1% dei sinistri include problemi di consenso informato (dati Centro GRC), probabilmente anche a causa di una sottovalutazione del diritto all'informazione da parte dei cittadini che si sottopongono a interventi programmati, connessa con una limitata consapevolezza del potere di scegliere le proprie cure che ogni persona dovrebbe poter esercitare posta di fronte alle opzioni terapeutiche disponibili per i propri problemi di salute","keywords":["Consenso informato","valutazione automatica della leggibilita\u0300","Trattamento Automatico del Linguaggio"],"pages":"35-39","url":"http:\/\/www.formas.toscana.it\/rivistadellasalute\/fileadmin\/files\/fascicoli\/2017\/212\/SeT_fascicolo_212.pdf","volume":"212","doi":"","editors":[],"editors_source":"","published":"SALUTE E TERRITORIO","publisher":"","issn":"0392-4505","isbn":"","conference_name":"","conference_place":"","conference_date":"","last_updated_cnr":"2025-06-12 01:43:35","last_updated_oai":"2025-06-12 01:43:35","last_updated_www":"0000-00-00 00:00:00"},{"id":1162,"id_source":342154,"institutes":["ILC"],"type":"conference_article","type_order":7,"title":"Identifying predictive features for textual genre classification: The key role of syntax","year":2017,"authors":["Cimino, A.","Wieling, M.","Dell'Orletta, F.","Montemagni, S.","Venturi, G."],"authors_source":"Cimino, A; Wieling, M; Dell'Orletta, F; Montemagni, S; Venturi, G","authors_cnr_name":["CIMINO, ANDREA","DELL'ORLETTA, FELICE","MONTEMAGNI, SIMONETTA","VENTURI, GIULIA"],"authors_cnr_id":["rp05770","rp22811","rp16780","rp00732"],"authors_cnr_institute":[],"abstract":"The paper investigates impact and role of different feature types for the specific task of Automatic Genre Classification with the final aim of identifying the most predictive ones. The goal was pursued by carrying out incremental feature selection through Grafting using different sets of linguistic features. Achieved results for discriminating among four traditional textual genres show the key role played by syntactic features, whose impact turned out to vary across genres","keywords":["Textual Genre Classification","Feature Selection","Syntactic Features"],"pages":"1-6","url":"http:\/\/www.scopus.com\/record\/display.url?eid=2-s2.0-85037370866&origin=inward","volume":"2006","doi":"","editors":[],"editors_source":"","published":"","publisher":"","issn":"","isbn":"","conference_name":"Italian Conference on Computational Linguistics (CLiC-it)","conference_place":"","conference_date":"","last_updated_cnr":"2025-06-14 00:47:04","last_updated_oai":"2025-06-14 00:47:04","last_updated_www":"0000-00-00 00:00:00"},{"id":1421,"id_source":372675,"institutes":["ILC"],"type":"journal_article","type_order":1,"title":"Monitoraggio linguistico di Scritture Brevi: aspetti metodologici e primi risultati","year":2016,"authors":["Brunato, D.","Dell'Orletta, F.","Montemagni, S.","Venturi, G."],"authors_source":"Brunato, D; Dell'Orletta, F; Montemagni, S; Venturi, G","authors_cnr_name":["BRUNATO, DOMINIQUE PIERINA","DELL'ORLETTA, FELICE","MONTEMAGNI, SIMONETTA","VENTURI, GIULIA"],"authors_cnr_id":["rp06836","rp22811","rp16780","rp00732"],"authors_cnr_institute":[],"abstract":"Se da un lato le tecnologie del linguaggio svolgono un ruolo ormai indiscusso per l'accesso al contenuto testuale, ci\u00f2 non appare scontato quando si va a considerare il loro ruolo nella valutazione delle strutture linguistiche sottostanti al testo. Questo contributo si focalizza sulla definizione di una metodologia innovativa di monitoraggio linguistico della lingua italiana che a partire dall'output di strumenti di annotazione linguistica automatica permette di ricostruire un profilo linguistico di una collezione di testi rappresentativa di una specifica variet\u00e0 d'uso della lingua. Tale metodologia \u00e8 stata applicata a un corpus di tweet allo scopo di far luce su interrogativi aperti quali la possibilit\u00e0 di rintracciare tendenze lessicali, morfo-sintattiche e sintattiche peculiari all'interno di questa tipologia testuale; di studiare come queste tendenze si rapportino ai tratti caratterizzanti della lingua scritta e parlata; di individuare possibili differenze nella forma linguistica in cui si twittano contenuti di natura diversa","keywords":["Trattamento Automatico del Linguaggio","Monitoraggio Linguistico","Varieta\u0300 d'Uso della Lingua","Lingua del Web"],"pages":"149-176","url":"https:\/\/iris.cnr.it\/handle\/20.500.14243\/372675","volume":"N. S. 5","doi":"","editors":[],"editors_source":"","published":"QUADERNI DI AI\u014cN","publisher":"","issn":"1825-2796","isbn":"","conference_name":"","conference_place":"","conference_date":"","last_updated_cnr":"2024-05-04 11:50:51","last_updated_oai":"2024-05-04 11:50:51","last_updated_www":"0000-00-00 00:00:00"},{"id":1586,"id_source":325822,"institutes":["ILC"],"type":"book_chapter","type_order":4,"title":"Le tecnologie linguistico-computazionali per la leggibilit\u00e0 della comunicazione istituzionale","year":2016,"authors":["Brunato, D.","Venturi, G."],"authors_source":"Brunato, Dominique; Venturi, Giulia","authors_cnr_name":["BRUNATO, DOMINIQUE PIERINA","VENTURI, GIULIA"],"authors_cnr_id":["rp06836","rp00732"],"authors_cnr_institute":[],"abstract":"Il contributo illustra il ruolo delle tecnologie linguistico-computazionali per la valutazione automatica della leggibilit\u00e0 dei testi della comunicazione istituzionale e propone alcuni esempi di semplificazione semi-automatica di testi amministrativi e normativi","keywords":["tecnologie linguistico-computazionali","valutazione automatica della leggibilita\u0300","comunicazione istituzionale"],"pages":"119-157","url":"https:\/\/iris.cnr.it\/handle\/20.500.14243\/325822","volume":"","doi":"","editors":[],"editors_source":"","published":"","publisher":"Pisa University Press (Pisa, ITA)","issn":"","isbn":"978-88-6741-627-1","conference_name":"","conference_place":"Pisa","conference_date":"","last_updated_cnr":"2024-07-09 11:29:57","last_updated_oai":"2024-07-09 11:29:57","last_updated_www":"0000-00-00 00:00:00"},{"id":1846,"id_source":351623,"institutes":["ILC"],"type":"edited_volume","type_order":5,"title":"Proceedings of the Workshop on Computational Linguistics for Linguistic Complexity (CL4LC 2016)","year":2016,"authors":["Brunato, D.","Dell'Orletta, F.","Venturi, G.","Fran\u00e7ois, T.","Blache, P."],"authors_source":"Dominique Brunato; Felice Dell'Orletta; Giulia Venturi; Thomas Fran\u00e7ois; Philippe Blache","authors_cnr_name":["BRUNATO, DOMINIQUE PIERINA","DELL'ORLETTA, FELICE","VENTURI, GIULIA"],"authors_cnr_id":["rp06836","rp22811","rp00732"],"authors_cnr_institute":[],"abstract":"Introduzione agli atti della prima edizione del workshop \"Computational Linguistics for Linguistic Complexity\" che raccoglie lavori che studiano da prospettive diverse il tema della complessit\u00e0 linguistica workshop allo scopo di promuovere una riflessione comune su approcci diversi all'indagine, al trattamento e alla valutazione di aspetti che rendono complessa la lingua","keywords":["Linguistic Complexity","Computational Linguistics"],"pages":"1-245","url":"https:\/\/aclweb.org\/anthology\/W\/W16\/W16-41.pdf","volume":"","doi":"","editors":[],"editors_source":"","published":"","publisher":"","issn":"","isbn":"978-4-87974-709-9","conference_name":"","conference_place":"","conference_date":"","last_updated_cnr":"2024-07-09 11:30:33","last_updated_oai":"2024-07-09 11:30:33","last_updated_www":"0000-00-00 00:00:00"},{"id":1217,"id_source":325812,"institutes":["ILC"],"type":"conference_article","type_order":7,"title":"CItA: an L1 Italian Learners Corpus to Study the Development of Writing Competence","year":2016,"authors":["Barbagli, A.","Lucisano, P.","Dell'Orletta, F.","Montemagni, S.","Venturi, G."],"authors_source":"Barbagli, A; Lucisano, P; Dell'Orletta, F; Montemagni, S; Venturi, G","authors_cnr_name":["DELL'ORLETTA, FELICE","MONTEMAGNI, SIMONETTA","VENTURI, GIULIA"],"authors_cnr_id":["rp22811","rp16780","rp00732"],"authors_cnr_institute":[],"abstract":"In this paper, we present the CItA corpus (Corpus Italiano di Apprendenti L1), a collection of essays written by Italian L1 learners collected during the first and second year of lower secondary school. The corpus was built in the framework of an interdisciplinary study jointly carried out by computational linguistics and experimental pedagogists and aimed at tracking the development of written language competence over the years and students' background information","keywords":["Italian Learner Corpus","Diachronic Evolution of Written Language Competence","Error Annotation"],"pages":"88-95","url":"http:\/\/www.lrec-conf.org\/proceedings\/lrec2016\/pdf\/536_Paper.pdf","volume":"","doi":"","editors":[],"editors_source":"","published":"","publisher":"European Language Resources Association ELRA (Paris, FRA)","issn":"","isbn":"978-2-9517408-9-1","conference_name":"Proceedings of the Tenth International Conference on Language Resources and Evaluation (LREC 2016)","conference_place":"Paris","conference_date":"","last_updated_cnr":"2024-04-14 10:33:28","last_updated_oai":"2024-04-14 10:33:28","last_updated_www":"0000-00-00 00:00:00"},{"id":1153,"id_source":333951,"institutes":["ILC"],"type":"conference_article","type_order":7,"title":"PaCCSS-IT: A Parallel Corpus of Complex-Simple Sentences for Automatic Text Simplification","year":2016,"authors":["Brunato, D.","Cimino, A.","Dell'Orletta, F.","Venturi, G."],"authors_source":"Brunato, Dominique; Cimino, Andrea; Dell'Orletta, Felice; Venturi, Giulia","authors_cnr_name":["BRUNATO, DOMINIQUE PIERINA","CIMINO, ANDREA","DELL'ORLETTA, FELICE","VENTURI, GIULIA"],"authors_cnr_id":["rp06836","rp05770","rp22811","rp00732"],"authors_cnr_institute":[],"abstract":"In this paper we present PaCCSS-IT, a Parallel Corpus of Complex-Simple Sentences for ITalian. To build the resource we develop a new method for automatically acquiring a corpus of complex-simple paired sentences able to intercept structural transformations and particularly suitable for text simplification. The method requires a wide amount of texts that can be easily extracted from the web making it suitable also for less-resourced languages. We test it on the Italian language making available the biggest Italian corpus for automatic text simplification","keywords":["Automatic Text Simplification","Sentence alignment","Italian corpus"],"pages":"351-361","url":"https:\/\/www.aclweb.org\/anthology\/D\/D16\/D16-1034.pdf","volume":"","doi":"10.18653\/v1\/d16-1034","editors":[],"editors_source":"","published":"","publisher":"Association for Computational Linguistics (Stroudsburg, USA)","issn":"","isbn":"978-1-945626-25-8","conference_name":"Conference on Empirical Methods in Natural Language Processing (EMNLP 2016)","conference_place":"Stroudsburg","conference_date":"","last_updated_cnr":"2025-06-14 00:46:18","last_updated_oai":"2025-06-14 00:46:18","last_updated_www":"0000-00-00 00:00:00"},{"id":1254,"id_source":325815,"institutes":["ILC"],"type":"conference_article","type_order":7,"title":"ULISSE: una strategia di adattamento al dominio per l'annotazione sintattica automatica","year":2016,"authors":["Dell'Orletta, F.","Venturi, G."],"authors_source":"Dell'Orletta, F; Venturi, G","authors_cnr_name":["DELL'ORLETTA, FELICE","VENTURI, GIULIA"],"authors_cnr_id":["rp22811","rp00732"],"authors_cnr_institute":[],"abstract":"This paper deals with Domain Adaptation for automatic syntactic annotation. Until the half of the 1980s, automatic linguistic annotation was based on algorithms built on groups of hand-written rules, defined a priori on the basis of the knowledge of the system to formalise. Subsequently, thanks to the progress of research in the field of Artificial Intelligence and to the development of linguistic resources, algorithms based on machine learning techniques began to be employed. The major difficulties of those algorithms were due to certain aspects of natural language such as ambiguities, diachronic evolutions, or language variations from the original domain of knowledge. More specifically, the issue of Domain Adaptation can be put in the following terms: \"can an annotated corpus [which is representative of a specific linguistic variety] be used for the syntactic analysis of a second corpus [which is representative of a different linguistic variety]?\". The author answer presenting an algorithm called ULISSE (Unsupervised LInguistically-driven Selection of dEpendency parses), which selects in an optima way the most representative sentences of a new target domain and feed them to the parser in addition to the original training set","keywords":["Domain Adaptation","annotazione sintattica automatica"],"pages":"55-79","url":"http:\/\/www.italianlp.it\/wp-content\/uploads\/2016\/10\/Compter_Parler_Soigner_ULISSE.pdf","volume":"","doi":"","editors":[],"editors_source":"","published":"","publisher":"","issn":"","isbn":"978-88-6952-038-9","conference_name":"Atti del convegno \"Compter parler soigner: tra linguistica e intelligenza artificiale\"","conference_place":"","conference_date":"","last_updated_cnr":"2024-04-14 10:51:22","last_updated_oai":"2024-04-14 10:51:22","last_updated_www":"0000-00-00 00:00:00"},{"id":1308,"id_source":325817,"institutes":["ILC"],"type":"conference_article","type_order":7,"title":"Dieci sfumature di marcatezza sintattica: Verso una nozione computazionale di complessita","year":2016,"authors":["Tusa, E.","Dell'Orletta, F.","Montemagni, S.","Venturi, G."],"authors_source":"Tusa E.; Dell'orletta F.; Montemagni S.; Venturi G.","authors_cnr_name":["DELL'ORLETTA, FELICE","MONTEMAGNI, SIMONETTA","VENTURI, GIULIA"],"authors_cnr_id":["rp22811","rp16780","rp00732"],"authors_cnr_institute":[],"abstract":"In this work, we will investigate whether and to what extent algorithms typically used to assess the reliability of the output of syntactic parsers can be used to study the correlation between processing complexity and the linguistic notion of markedness. Although still preliminary, achieved results show the key role of features such as dependency direction and length in defining the markedness degrees of a given syntactic construction","keywords":["marcatezza sintattica","complessita\u0300 linguistica","annotazione linguistica automatica"],"pages":"1-6","url":"http:\/\/www.scopus.com\/record\/display.url?eid=2-s2.0-85009279517&origin=inward","volume":"1749","doi":"","editors":[],"editors_source":"","published":"","publisher":"","issn":"","isbn":"","conference_name":"Italian Conference on Computational Linguistics (CLiC-it)","conference_place":"","conference_date":"","last_updated_cnr":"2025-06-14 00:50:46","last_updated_oai":"2025-06-14 00:50:46","last_updated_www":"0000-00-00 00:00:00"},{"id":2056,"id_source":322610,"institutes":["ILC"],"type":"journal_article","type_order":1,"title":"Il ruolo delle tecnologie del linguaggio nel monitoraggio dell'evoluzione delle abilit\u00e0 di scrittura: primi risultati","year":2015,"authors":["Barbagli, A.","Lucisano, P.","Dell'Orletta, F.","Montemagni, S.","Venturi, G."],"authors_source":"Barbagli A.; Lucisano P.; Dell'Orletta F.; Montemagni S.; Venturi G.","authors_cnr_name":["DELL'ORLETTA, FELICE","MONTEMAGNI, SIMONETTA","VENTURI, GIULIA"],"authors_cnr_id":["rp22811","rp16780","rp00732"],"authors_cnr_institute":[],"abstract":"L'ultimo decennio ha visto l'affermarsi a livello internazionale dell'uso di tecnologie del linguaggio per lo studio dei processi di apprendimento. Questo contributo riporta i primi e promettenti risultati di uno studio interdisciplinare che si \u00e8 avvalso di metodi e tecniche di analisi propri della linguistica computazionale, della linguistica e della pedagogia sperimentale. Lo studio, finalizzato al monitoraggio dell'evoluzione del processo di apprendimento della lingua italiana, \u00e8 stato condotto a partire dalle produzione scritte di studenti della scuola secondaria di primo grado con strumenti di annotazione linguistica automatica e di estrazione di conoscenza e ha portato all'identificazione di un insieme di tratti qualificanti il processo di apprendimento linguistico","keywords":["evoluzione delle competenze linguistiche","Didattica Sperimentale","Estrazione di conoscenza","Annotazione linguistica automatica"],"pages":"99-117","url":"https:\/\/journals.openedition.org\/ijcol\/326","volume":"","doi":"10.4000\/ijcol.326","editors":[],"editors_source":"","published":"IJCOL","publisher":"","issn":"2499-4553","isbn":"","conference_name":"","conference_place":"","conference_date":"","last_updated_cnr":"2024-06-09 13:18:21","last_updated_oai":"2024-06-09 13:18:21","last_updated_www":"0000-00-00 00:00:00"},{"id":2355,"id_source":340887,"institutes":["ILC"],"type":"book_chapter","type_order":4,"title":"Neuroscienze e genetica comportamentale in un corpus di sentenze italiane alla luce dei risultati di elaborazioni linguistico-computazionali","year":2015,"authors":["Sagri, M.","Tiscornia, D.","Venturi, G.","Montemagni, S."],"authors_source":"Sagri, Mt; Tiscornia, D; Venturi, G; Montemagni, S","authors_cnr_name":["SAGRI, MARIA TERESA","TISCORNIA, DANIELA","VENTURI, GIULIA","MONTEMAGNI, SIMONETTA"],"authors_cnr_id":["rp17388","rp22053","rp00732","rp16780"],"authors_cnr_institute":[],"abstract":"Il contributo intende illustrare i primi risultati di uno studio, tutt'ora in corso, finalizzato a monitorare evoluzione e mutamenti dell'uso della neurogenetica e delle neuroscienze all'interno del sistema della giustizia in Italia","keywords":[],"pages":"","url":"https:\/\/iris.cnr.it\/handle\/20.500.14243\/340887","volume":"","doi":"","editors":[],"editors_source":"","published":"","publisher":"","issn":"","isbn":"9788849529326","conference_name":"","conference_place":"","conference_date":"","last_updated_cnr":"2024-05-15 14:42:37","last_updated_oai":"0000-00-00 00:00:00","last_updated_www":"0000-00-00 00:00:00"},{"id":537,"id_source":322147,"institutes":["ILC"],"type":"conference_article","type_order":7,"title":"CItA: un Corpus di Produzioni Scritte di Apprendenti l'Italiano L1 Annotato con Errori","year":2015,"authors":["Barbagli, A.","Lucisano, P.","Dell'Orletta, F.","Montemagni, S.","Venturi, G."],"authors_source":"Barbagli, Alessia; Lucisano, Pietro; Dell'Orletta, Felice; Montemagni, Simonetta; Venturi, Giulia","authors_cnr_name":["DELL'ORLETTA, FELICE","MONTEMAGNI, SIMONETTA","VENTURI, GIULIA"],"authors_cnr_id":["rp22811","rp16780","rp00732"],"authors_cnr_institute":[],"abstract":"In questo articolo presentiamo CItA il primo corpus di produzioni scritte di apprendenti l'italiano L1 del primo e del secondo anno della scuola secondaria di primo grado annotato con errori grammaticali, ortografici e lessicali. Le specificit\u00e0 del corpus e la sua natura diacronica lo rendono particolarmente utile sia per applicazioni linguistico-computazionali sia per studi socio-pedagogici","keywords":["Apprendiemento della lingua madre","evoluzione delle competenze linguistiche"],"pages":"31-35","url":"http:\/\/www.italianlp.it\/wp-content\/uploads\/2016\/03\/CItA_errori.pdf","volume":"","doi":"","editors":[],"editors_source":"","published":"","publisher":"Accademia University Press (Torino, ITA)","issn":"","isbn":"978-88-99200-62-6","conference_name":"2nd Italian Conference on Computational Linguistics (CLiC-it)","conference_place":"Torino","conference_date":"","last_updated_cnr":"2024-05-15 14:38:17","last_updated_oai":"2024-05-15 14:38:17","last_updated_www":"0000-00-00 00:00:00"},{"id":1827,"id_source":296574,"institutes":["ILC"],"type":"conference_article","type_order":7,"title":"Design and Annotation of the First Italian Corpus for Text Simplification","year":2015,"authors":["Brunato, D.","Dell'Orletta, F.","Venturi, G.","Montemagni, S."],"authors_source":"Brunato D.; Dell'Orletta F.; Venturi G.; Montemagni S.","authors_cnr_name":["BRUNATO, DOMINIQUE PIERINA","DELL'ORLETTA, FELICE","VENTURI, GIULIA","MONTEMAGNI, SIMONETTA"],"authors_cnr_id":["rp06836","rp22811","rp00732","rp16780"],"authors_cnr_institute":[],"abstract":"In this paper, we present design and construction of the first Italian corpus for automatic and semi-automatic text simplification. In line with current approaches, we propose a new annotation scheme specifically conceived to identify the typology of changes an original sentence undergoes when it is manually simplified. Such a scheme has been applied to two aligned Italian corpora, containing original texts with corresponding simplified versions, selected as representative of two different manual simplification strategies and addressing different target reader populations. Each corpus was annotated with the operations foreseen in the annotation scheme, covering different levels of linguistic description. Annotation results were analysed with the final aim of capturing peculiarities and differences of the different simplification strategies pursued in the two corpora","keywords":["Annotation Scheme","Automatic Text Simplification"],"pages":"31-34","url":"https:\/\/aclweb.org\/anthology\/W\/W15\/W15-1604.pdf","volume":"","doi":"","editors":[],"editors_source":"","published":"","publisher":"","issn":"","isbn":"978-1-941643-47-1","conference_name":"Proceedings of LAW IX-The 9th Linguistic Annotation Workshop","conference_place":"","conference_date":"","last_updated_cnr":"2024-04-26 22:57:26","last_updated_oai":"2024-04-26 22:57:26","last_updated_www":"0000-00-00 00:00:00"},{"id":701,"id_source":322145,"institutes":["ILC"],"type":"conference_article","type_order":7,"title":"Tracking the Evolution of Written Language Competence: an NLP-based Approach","year":2015,"authors":["Richter, S.","Cimino, A.","Dell'Orletta, F.","Venturi, G."],"authors_source":"Richter S.; Cimino A.; Dell'Orletta F.; Venturi G.","authors_cnr_name":["CIMINO, ANDREA","DELL'ORLETTA, FELICE","VENTURI, GIULIA"],"authors_cnr_id":["rp05770","rp22811","rp00732"],"authors_cnr_institute":[],"abstract":"In this paper, we present an NLP-based innovative approach for tracking the evolution of written language competence relying on different sets of linguistic features that predict text quality. This approach was tested on a corpus essays written by Italian L1 learners of the first and second year of the lower secondary school","keywords":["Evolution of Written Language Competence","multi-level linguistic analysis"],"pages":"236-240","url":"http:\/\/www.italianlp.it\/wp-content\/uploads\/2016\/03\/tracking-language-competence.pdf","volume":"","doi":"","editors":[],"editors_source":"","published":"","publisher":"Accademia University Press (Torino, ITA)","issn":"","isbn":"978-88-99200-62-6","conference_name":"2nd Italian Conference on Computational Linguistics (CLiC-it)","conference_place":"Torino","conference_date":"","last_updated_cnr":"2024-05-12 00:45:10","last_updated_oai":"2024-05-12 00:45:10","last_updated_www":"0000-00-00 00:00:00"},{"id":1943,"id_source":304237,"institutes":["ILC"],"type":"conference_article","type_order":7,"title":"NLP-Based Readability Assessment of Health-Related Texts: a Case Study on Italian Informed Consent Forms","year":2015,"authors":["Venturi, G.","Bellandi, T.","Dell'Orletta, F.","Montemagni, S."],"authors_source":"Giulia Venturi; Tommaso Bellandi; Felice Dell'Orletta; Simonetta Montemagni","authors_cnr_name":["VENTURI, GIULIA","DELL'ORLETTA, FELICE","MONTEMAGNI, SIMONETTA"],"authors_cnr_id":["rp00732","rp22811","rp16780"],"authors_cnr_institute":[],"abstract":"The paper illustrates the results of a case study aimed at investigating and enhancing the accessibility of Italian health-related documents by relying on advanced NLP techniques, with particular attention to informed consent forms. Results achieved show that the features automatically extracted from the linguistically annotated text and ranging across different levels of linguistic description have a high discriminative power in order to guarantee a reliable readability assessment","keywords":["Readability assessment","health-related information"],"pages":"131-141","url":"http:\/\/www.aclweb.org\/anthology\/W15-2618","volume":"","doi":"","editors":[],"editors_source":"","published":"","publisher":"","issn":"","isbn":"978-1-941643-32-7","conference_name":"Sixth International Workshop on Health Text Mining and Information Analysis (Louhi)","conference_place":"","conference_date":"","last_updated_cnr":"2024-05-25 11:35:18","last_updated_oai":"2024-05-25 11:35:18","last_updated_www":"0000-00-00 00:00:00"},{"id":466,"id_source":304238,"institutes":["ILC"],"type":"conference_misc","type_order":8,"title":"Language technologies for automatic readability assessment of health-related Information: a preliminary investigation into the informed consent forms used in a regional health service","year":2015,"authors":["Venturi, G.","Rinnone, S.","Montemagni, S.","Sassi, M.","Terranova, G.","Flore, E.","Bellandi, T."],"authors_source":"Giulia Venturi; Sabrina Rinnone; Simonetta Montemagni; Manuela Sassi; Giuseppina Terranova; Elisabetta Flore; Tommaso Bellandi","authors_cnr_name":["VENTURI, GIULIA","MONTEMAGNI, SIMONETTA","SASSI, MANUELA"],"authors_cnr_id":["rp00732","rp16780","rp21726"],"authors_cnr_institute":[],"abstract":"Rationale: Within an information society, where everyone should be able to access all available information, improving access to written language is becoming more and more a central issue. This is the case for health-related information which should be accessible to all members of the society, including people who have reading difficulties as a result of a low education level or of language-based learning disabilities or because the language of the text is not their native language. Moreover, the breakdown of doctor-patient communication is one of the most frequent cause of adverse events. Research questions: We conducted a preliminary investigation to assess the readability of a corpus of informed consent forms used before a clinical procedure in the hospitals of a Regional Healthcare Service. Secondary goals include the comparison of readability across specialties and healthcare trusts. Methods: Providing complex scientific information in a way that is comprehensible to a lay person is a challenge that nowadays can be addressed by resorting to advanced Natural Language Processing (NLP) techniques, which make it possible to monitor the linguistic complexity of texts at the syntactic and lexical levels and to support their simplification, whenever needed. The study has been carried out by combining NLP-enabled feature extraction and state-of-the-art machine learning algorithms. To this end we used READ-IT, the first NLP-based readability assessment tool for Italian. Results: We analysed 584 documents, covering 29 specialties, for a total of 607. 790 word tokens, currently used at the 36 public hospitals in Tuscany. Although the readability level of all documents in the corpus is low, both at the lexical and syntactic level, significant differences can be observed between specialties and healthcare trust releasing the forms. With the readability level ranging between 0 (easy-to-read) and 100 (difficult-to-read), it resulted that the pediatric informed consent documents are the most easy-to-read forms (with an average score of 75) while the most difficult-to read documents are documents of the surgical area (whose average score is 80) (standard deviation 2). Discussion: The state of the art resulting from this preliminary study shows that NLP-based readability assessment tools can help to measure the linguistic complexity of informed consent forms and guide the editor to identify linguistically complex passages that need to be simplified, either syntactically or lexically. The use of an assessment tool designed for the general language is the main limitation of the study and should be addressed through the customization of the tool to assess the readability of the healthcare jargon. A further step of the research consider also the design of a guidance to prepare readable informed consent forms","keywords":["Readability assessment","health-related information"],"pages":"","url":"http:\/\/static1.squarespace.com\/static\/561c0d01e4b0b5ad2e65cc48\/t\/561d44dfe4b089431662d174\/1444758751213\/LibrettoProgramma.pdf","volume":"","doi":"","editors":[],"editors_source":"","published":"","publisher":"","issn":"","isbn":"","conference_name":"ISCOME 2015 Conference: \"The Golden Bridge: Communication and Patient Safety\"","conference_place":"","conference_date":"","last_updated_cnr":"2024-04-27 23:17:55","last_updated_oai":"2024-04-27 23:17:55","last_updated_www":"0000-00-00 00:00:00"},{"id":1293,"id_source":245146,"institutes":["ILC"],"type":"journal_article","type_order":1,"title":"Le tecnologie linguistico-computazionali nella misura della leggibilit\u00e0 di testi giuridici","year":2014,"authors":["Brunato, D.","Venturi, G."],"authors_source":"Brunato, D; Venturi, G","authors_cnr_name":["BRUNATO, DOMINIQUE PIERINA","VENTURI, GIULIA"],"authors_cnr_id":["rp06836","rp00732"],"authors_cnr_institute":[],"abstract":"This paper presents an innovative NLP-based methodology for automatically assessing the readability of legal documents, with a view to their simplification. As part of a broader research field investigating the accessibility of the language of the law, we consider the specific domain of the bureaucratic language through an analysis of real texts; the accessibility to this typology of texts is indeed a crucial requisite of the communication between institutions and citizens. To the best of our knowledge, this is the first attempt aimed at showing that state-of-the-art language technologies are nowadays mature not only to enable the automatic readability evaluation of legal texts but also to support their simplification. For these purposes, we adopted READ-IT, the first advanced readability assessment tool for the Italian language","keywords":[],"pages":"111-142","url":"https:\/\/iris.cnr.it\/handle\/20.500.14243\/245146","volume":"XXIII (1)","doi":"","editors":[],"editors_source":"","published":"INFORMATICA E DIRITTO","publisher":"","issn":"0390-0975","isbn":"","conference_name":"","conference_place":"","conference_date":"","last_updated_cnr":"2024-05-14 20:51:10","last_updated_oai":"2024-05-14 20:51:10","last_updated_www":"0000-00-00 00:00:00"},{"id":5,"id_source":260898,"institutes":["ILC"],"type":"journal_article","type_order":1,"title":"Assessing document and sentence readability in less resourced languages and across textual genres","year":2014,"authors":["Dell'Orletta, F.","Montemagni, S.","Venturi, G."],"authors_source":"Felice Dell'Orletta; Simonetta Montemagni; Giulia Venturi","authors_cnr_name":["DELL'ORLETTA, FELICE","MONTEMAGNI, SIMONETTA","VENTURI, GIULIA"],"authors_cnr_id":["rp22811","rp16780","rp00732"],"authors_cnr_institute":[],"abstract":"In this paper, we tackle three underresearched issues of the automatic readability assessment literature, namely the evaluation of text readability in less resourced languages, with respect to sentences (as opposed to documents) as well as across textual genres. Different solutions to these issues have been tested by using and refining READ-IT, the first advanced readability assessment tool for Italian, which combines traditional raw text features with lexical, morpho-syntactic and syntactic information. In READ-IT readability assessment is carried out with respect to both documents and sentences, with the latter constituting an important novelty of the proposed approach: READ-IT shows a high accuracy in the document classification task and promising results in the sentence classification scenario. By comparing the results of two versions of READ-IT, adopting a classification-versus ranking-based approach, we also show that readability assessment is strongly influenced by textual genre; for this reason a genre-oriented notion of readability is needed. With classification-based approaches, reliable results can only be achieved with genre-specific models: Since this is far from being a workable solution, especially for less resourced languages, a new ranking method for readability assessment is proposed, based on the notion of distance","keywords":["readability assessment","less resourced languages","multi-level linguistic annotation","textual genres"],"pages":"163-193","url":"http:\/\/www.ingentaconnect.com\/content\/jbp\/itl\/2014\/00000165\/00000002\/art00005","volume":"165 (2)","doi":"10.1075\/itl.165.2.03del","editors":[],"editors_source":"","published":"","publisher":"","issn":"","isbn":"","conference_name":"","conference_place":"","conference_date":"","last_updated_cnr":"2024-03-24 20:16:49","last_updated_oai":"2024-03-24 20:16:49","last_updated_www":"0000-00-00 00:00:00"},{"id":897,"id_source":280049,"institutes":["ILC","ITTIG","IGSG"],"type":"edited_volume","type_order":5,"title":"Proceedings of the Fourth Workshop on Semantic Processing of Legal Texts","year":2014,"authors":["Francesconi, E.","Montemagni, S.","Peters, W.","Venturi, G.","Wyner, A."],"authors_source":"Francesconi, E; Montemagni, S; Peters, W; Venturi, G; Wyner, A","authors_cnr_name":["FRANCESCONI, ENRICO","MONTEMAGNI, SIMONETTA","VENTURI, GIULIA"],"authors_cnr_id":["rp03980","rp16780","rp00732"],"authors_cnr_institute":[],"abstract":"","keywords":[],"pages":"33","url":"http:\/\/www.lrec-conf.org\/proceedings\/lrec2014\/workshops\/LREC2014Workshop-SPLeT%20Proceedings.pdf","volume":"","doi":"","editors":[],"editors_source":"","published":"","publisher":"","issn":"","isbn":"978-2-9517408-8-4","conference_name":"","conference_place":"Parigi","conference_date":"","last_updated_cnr":"2024-04-09 11:18:20","last_updated_oai":"2024-04-09 11:18:20","last_updated_www":"0000-00-00 00:00:00"},{"id":1570,"id_source":266268,"institutes":["ILC"],"type":"conference_article","type_order":7,"title":"Tecnologie del linguaggio e monitoraggio dell'evoluzione delle abilit\u00e0 di scrittura nella scuola secondaria di primo grado","year":2014,"authors":["Barbagli, A.","Lucisano, P.","Dell'Orletta, F.","Montemagni, S.","Venturi, G."],"authors_source":"Barbagli, A; Lucisano, P; Dell'Orletta, F; Montemagni, S; Venturi, G","authors_cnr_name":["DELL'ORLETTA, FELICE","MONTEMAGNI, SIMONETTA","VENTURI, GIULIA"],"authors_cnr_id":["rp22811","rp16780","rp00732"],"authors_cnr_institute":[],"abstract":"Over the last ten years, the use of language technologies was successfully extended to the study of learning processes. The paper reports the first results of a study, which is part of a broader experimental pedagogy project, aimed at monitoring the evolution of the learning process of the Italian language based on a corpus of written productions by students and exploiting automatic linguistic annotation and knowledge extraction tools","keywords":[],"pages":"23-27","url":"http:\/\/www.italianlp.it\/wp-content\/uploads\/2014\/12\/Tecnologie-del-linguaggio-per-la-scuola.pdf","volume":"","doi":"10.12871\/CLICIT201415","editors":["Basili, R.","Lenci, A.","Magnini, B."],"editors_source":"Roberto Basili, Alessandro Lenci, Bernardo Magnini","published":"Proceedings of the First Italian Conference on Computational Linguistics (CLiC-it 2014)","publisher":"Pisa University Press srl (Pisa, ITA)","issn":"","isbn":"978-8-86741-472-7","conference_name":"First Italian Conference on Computational Linguistics (CLiC-it 2014)","conference_place":"Pisa","conference_date":"","last_updated_cnr":"2024-04-17 11:53:24","last_updated_oai":"2024-04-17 11:53:24","last_updated_www":"0000-00-00 00:00:00"},{"id":1427,"id_source":228541,"institutes":["ILC"],"type":"conference_article","type_order":7,"title":"Computational Analysis of Historical Documents: An Application to Italian War Bulletins in World War I and II","year":2014,"authors":["Boschetti, F.","Cimino, A.","Dell'Orletta, F.","Lebani, G. E.","Passaro, L.","Picchi, P.","Venturi, G.","Montemagni, S. L. A."],"authors_source":"Boschetti F.; Cimino A.; Dell'Orletta F.; Lebani G.E.; Passaro L.; Picchi P.; Venturi G.; Montemagni S. Lenci A.","authors_cnr_name":["BOSCHETTI, FEDERICO","CIMINO, ANDREA","DELL'ORLETTA, FELICE","PICCHI, PAOLO","VENTURI, GIULIA"],"authors_cnr_id":["rp04876","rp05770","rp22811","rp11135","rp00732"],"authors_cnr_institute":[],"abstract":"World War (WW) I and II represent crucial landmarks in the history on mankind: They have affected the destiny of whole generations and their consequences are still alive throughout Europe. In this paper we present an ongoing project to carry out a computational analysis of Italian war bulletins in WWI and WWII, by applying state-of-the-art tools for NLP and Information Extraction. The annotated texts and extracted information will be explored with a dedicated Web interface, allowing for multidimensional access and exploration of historical events through space and time","keywords":["World War I"],"pages":"70-75","url":"http:\/\/www.lrec-conf.org\/proceedings\/lrec2014\/workshops\/LREC2014Workshop-LRT4HDA%20Proceedings.pdf","volume":"","doi":"","editors":[],"editors_source":"","published":"Proceedings of workshop on Language resources and technologies for processing and linking historical documents and archives-Deploying Linked Open Data in Cultural Heritage-LREC 2014, 26 May, Reykjavik, Iceland","publisher":"European language resources association (ELRA) (Paris, FRA)","issn":"","isbn":"","conference_name":"LREC 2014","conference_place":"Paris","conference_date":"","last_updated_cnr":"2024-05-12 11:22:31","last_updated_oai":"2024-05-12 11:22:31","last_updated_www":"0000-00-00 00:00:00"},{"id":48,"id_source":266263,"institutes":["ILC"],"type":"conference_article","type_order":7,"title":"Defining an annotation scheme with a view to automatic text simplification","year":2014,"authors":["Brunato, D.","Dell'Orletta, F.","Venturi, G.","Montemagni, S."],"authors_source":"Brunato D.; Dell'Orletta F.; Venturi G.; Montemagni S.","authors_cnr_name":["BRUNATO, DOMINIQUE PIERINA","DELL'ORLETTA, FELICE","VENTURI, GIULIA","MONTEMAGNI, SIMONETTA"],"authors_cnr_id":["rp06836","rp22811","rp00732","rp16780"],"authors_cnr_institute":[],"abstract":"This paper presents the preliminary steps of ongoing research in the field of automatic text simplification. In line with current approaches, we propose here a new annotation scheme specifically conceived to identify the typologies of changes an original sentence undergoes when it is manually simplified. Such a scheme has been tested on a parallel corpus available for Italian, which we have first aligned at sentence level and then annotated with simplification rules","keywords":[],"pages":"87-92","url":"http:\/\/www.italianlp.it\/wp-content\/uploads\/2014\/12\/Text-simplification.pdf","volume":"","doi":"10.12871\/CLICIT2014118","editors":["Basili, R.","Lenci, A.","Magnini, B."],"editors_source":"Roberto Basili, Alessandro Lenci, Bernardo Magnini","published":"Proceedings of the First Italian Conference on Computational Linguistics (CLiC-it 2014)","publisher":"Pisa University Press srl (Pisa, ITA)","issn":"","isbn":"978-8-86741-472-7","conference_name":"First Italian Conference on Computational Linguistics (CLiC-it 2014)","conference_place":"Pisa","conference_date":"","last_updated_cnr":"2024-06-14 13:06:06","last_updated_oai":"2024-06-14 13:06:06","last_updated_www":"0000-00-00 00:00:00"},{"id":1117,"id_source":226944,"institutes":["ILC"],"type":"conference_article","type_order":7,"title":"T2K: a System for Automatically Extracting and Organizing Knowledge from Texts","year":2014,"authors":["Dell'Orletta, F.","Venturi, G.","Cimino, A.","Montemagni, S."],"authors_source":"Felice Dell'Orletta; Giulia Venturi; Andrea Cimino; Simonetta Montemagni","authors_cnr_name":["DELL'ORLETTA, FELICE","VENTURI, GIULIA","CIMINO, ANDREA","MONTEMAGNI, SIMONETTA"],"authors_cnr_id":["rp22811","rp00732","rp05770","rp16780"],"authors_cnr_institute":[],"abstract":"In this paper, we present T2K, a suite of tools for automatically extracting domain-specific knowledge from collections of Italian and English texts. T2K (Text-To-Knowledge v2) relies on a battery of tools for Natural Language Processing (NLP), statistical text analysis and machine learning which are dynamically integrated to provide an accurate and incremental representation of the content of vast repositories of unstructured documents. Extracted knowledge ranges from domain-specific entities and named entities to the relations connecting them and can be used for indexing document collections with respect to different information types. T2K also includes \"linguistic profiling\" functionalities aimed at supporting the user in constructing the acquisition corpus, e. g. in selecting texts belonging to the same genre or characterized by the same degree of specialization or in monitoring the \"added value\" of newly inserted documents. T2K is a web application which can be accessed from any browser through a personal account which has been tested in a wide range of domains","keywords":["Natural Language Processing","Information Extraction","Knowledge Management"],"pages":"2062-2070","url":"http:\/\/www.lrec-conf.org\/proceedings\/lrec2014\/pdf\/590_Paper.pdf","volume":"","doi":"","editors":[],"editors_source":"","published":"","publisher":"","issn":"","isbn":"978-2-9517408-8-4","conference_name":"International Conference on Language Resources and Evaluation (LREC)","conference_place":"","conference_date":"","last_updated_cnr":"2025-06-14 00:38:30","last_updated_oai":"2025-06-14 00:38:30","last_updated_www":"0000-00-00 00:00:00"},{"id":1415,"id_source":266274,"institutes":["ILC"],"type":"conference_article","type_order":7,"title":"Assessing the readability of sentences: which corpora and features?","year":2014,"authors":["Dell'Orletta, F.","Wieling, M.","Cimino, A.","Venturi, G.","Montemagni, S."],"authors_source":"Dell'Orletta F.; Wieling M.; Cimino A.; Venturi G.; Montemagni S.","authors_cnr_name":["DELL'ORLETTA, FELICE","CIMINO, ANDREA","VENTURI, GIULIA","MONTEMAGNI, SIMONETTA"],"authors_cnr_id":["rp22811","rp05770","rp00732","rp16780"],"authors_cnr_institute":[],"abstract":"The paper investigates the problem of sentence readability assessment, which is modelled as a classification task, with a specific view to text simplification. In particular, it addresses two open issues connected with it, i. e. the corpora to be used for training, and the identification of the most effective features to determine sentence readability. An existing readability assessment tool developed for Italian was specialized at the level of training corpus and learning algorithm. A maximum entropy-based feature selection and ranking algorithm (grafting) was used to identify to the most relevant features: it turned out that assessing the readability of sentences is a complex task, requiring a high number of features, mainly syntactic ones","keywords":[],"pages":"163-173","url":"http:\/\/acl2014.org\/acl2014\/W14-18\/pdf\/W14-1820.pdf","volume":"","doi":"","editors":[],"editors_source":"","published":"Proceedings of 9th Workshop on Innovative Use of NLP for Building Educational Applications (BEA 2014)","publisher":"Association for Computational Linguistics (Stroudsburg, USA)","issn":"","isbn":"978-1-941643-03-7","conference_name":"9th Workshop on Innovative Use of NLP for Building Educational Applications (BEA 2014)","conference_place":"Stroudsburg","conference_date":"","last_updated_cnr":"2024-05-12 00:42:24","last_updated_oai":"2024-05-12 00:42:24","last_updated_www":"0000-00-00 00:00:00"},{"id":94,"id_source":289336,"institutes":["ILC","ITTIG","IGSG"],"type":"conference_misc","type_order":8,"title":"Investigating the relationship between neuroscience and law: a case study on a corpus of Italian case law texts","year":2014,"authors":["Sagri, M. T.","Tiscornia, D.","Montemagni, S.","Venturi, G."],"authors_source":"M.T.Sagri; D. Tiscornia; S. Montemagni; G. Venturi;","authors_cnr_name":["TISCORNIA, DANIELA","MONTEMAGNI, SIMONETTA","VENTURI, GIULIA"],"authors_cnr_id":["rp22053","rp16780","rp00732"],"authors_cnr_institute":[],"abstract":"","keywords":["Neuroscience","linguistic and lexico-semantic analysis"],"pages":"","url":"https:\/\/iris.cnr.it\/handle\/20.500.14243\/289336","volume":"","doi":"","editors":[],"editors_source":"","published":"","publisher":"","issn":"","isbn":"","conference_name":"Language and Law in Social Practice 3rd International Conference","conference_place":"","conference_date":"","last_updated_cnr":"2024-06-21 13:01:42","last_updated_oai":"2024-06-21 13:01:42","last_updated_www":"0000-00-00 00:00:00"},{"id":1589,"id_source":280032,"institutes":["ILC"],"type":"journal_article","type_order":1,"title":"Linguistically-driven selection of correct arcs for dependency parsing","year":2013,"authors":["Dell'Orletta, F.","Venturi, G.","Montemagni, S."],"authors_source":"Dell'Orletta, F; Venturi, G; Montemagni, S","authors_cnr_name":["DELL'ORLETTA, FELICE","VENTURI, GIULIA","MONTEMAGNI, SIMONETTA"],"authors_cnr_id":["rp22811","rp00732","rp16780"],"authors_cnr_institute":[],"abstract":"LISCA is an unsupervised algorithm aimed at assigning a quality score to each arc generated by a dependency parser in order to produce a decreasing ranking of arcs from correct to incorrect ones. LISCA exploits statistics about a set of linguistically-motivated and dependency-based features extracted from a large corpus of automatically parsed sentences and uses them to assign a quality score to each arc of a parsed sentence belonging to the same domain of the automatically parsed corpus. LISCA has been successfully tested on two datasets belonging to two different domains and in all experiments it turned out to outperform different baselines, thus showing to be able to reliably detect correct arcs also representing domain-specific peculiarities","keywords":["Correct arcs","Dependency parsing"],"pages":"125-136","url":"http:\/\/cys.cic.ipn.mx\/ojs\/index.php\/CyS\/article\/view\/1517","volume":"17 (2)","doi":"","editors":[],"editors_source":"","published":"COMPUTACI\u00d3N Y SISTEMAS","publisher":"","issn":"1405-5546","isbn":"","conference_name":"","conference_place":"","conference_date":"","last_updated_cnr":"2024-04-28 10:42:19","last_updated_oai":"2024-04-28 10:42:19","last_updated_www":"0000-00-00 00:00:00"},{"id":2023,"id_source":287820,"institutes":["ILC"],"type":"book_chapter","type_order":4,"title":"Evalita 2011: the Frame Labeling over Italian Texts Task","year":2013,"authors":["Basili, R.","Lenci, A.","De Cao, D.","Moschitti, A.","Venturi, G."],"authors_source":"Basili R; Lenci A; De Cao D; Moschitti A; Venturi G","authors_cnr_name":["VENTURI, GIULIA"],"authors_cnr_id":["rp00732"],"authors_cnr_institute":[],"abstract":"The Frame Labeling over Italian Texts (FLaIT) task held within the EvalIta 2011 challenge is here described. It focuses on the automatic annotation of free texts according to frame semantics. Systems were asked to label all semantic frames and their arguments, as evoked by predicate words occurring in plain text sentences. Proposed systems are based on a variety of learning techniques and achieve very good results, over 80% of accuracy, in most subtasks","keywords":["NLP System Evaluation","Shallow Semantic Parsing","Frame Semantics"],"pages":"195-204","url":"https:\/\/iris.cnr.it\/handle\/20.500.14243\/287820","volume":"","doi":"","editors":["Magnini, B.","Cutugno, F.","Falcone, M.","Pianta, E."],"editors_source":"Bernardo Magnini, Francesco Cutugno, Mauro Falcone, Emanuele Pianta","published":"Evaluation of Natural Language and Speech Tools for Italian","publisher":"Springer (Berlin Heidelberg, DEU)","issn":"","isbn":"978-3-642-35827-2","conference_name":"","conference_place":"Berlin Heidelberg","conference_date":"","last_updated_cnr":"2025-06-15 00:24:37","last_updated_oai":"2025-06-15 00:24:37","last_updated_www":"0000-00-00 00:00:00"},{"id":450,"id_source":220446,"institutes":["ILC","ITTIG","IGSG"],"type":"book_chapter","type_order":4,"title":"Domain Adaptation for Dependency Parsing at EVALITA 2011","year":2013,"authors":["Dell'Orletta, F.","Marchi, S.","Montemagni, S.","Venturi, G.","Agnoloni, T.","Francesconi, E."],"authors_source":"F. Dell'Orletta;S. Marchi;S. Montemagni;G. Venturi;T. Agnoloni;E. Francesconi","authors_cnr_name":["DELL'ORLETTA, FELICE","MARCHI, SIMONE","MONTEMAGNI, SIMONETTA","VENTURI, GIULIA","AGNOLONI, TOMMASO","FRANCESCONI, ENRICO"],"authors_cnr_id":["rp22811","rp09286","rp16780","rp00732","rp21506","rp03980"],"authors_cnr_institute":[],"abstract":"The domain adaptation task was aimed at investigating techniques for adapting state-of-the-art dependency parsing systems to new domains. Both the language dealt with, i. e. Italian, and the target do-main, namely the legal domain, represent two main novelties of the task organised at Evalita 2011 with respect to previous domain adaptation ini-tiatives. In this paper, we define the task and describe how the datasets were created from different resources. In addition, we characterize the different approaches of the participating systems, report the test results, and provide a first analysis of these results","keywords":["Dependency Parsing","Domain Adaptation","Self-training","Active Learning","Legal-NLP"],"pages":"58-69","url":"https:\/\/iris.cnr.it\/handle\/20.500.14243\/220446","volume":"","doi":"","editors":["Magnini, B.","Cutugno, F.","Falcone, M.","Pianta, E."],"editors_source":"Bernardo Magnini, Francesco Cutugno, Mauro Falcone, Emanuele Pianta","published":"Evaluation of NLP and Speech Tools for Italian","publisher":"Springer (Berlin Heidelberg, DEU)","issn":"","isbn":"978-3-642-35827-2","conference_name":"","conference_place":"Berlin Heidelberg","conference_date":"","last_updated_cnr":"2024-03-11 03:34:27","last_updated_oai":"2024-03-11 03:34:27","last_updated_www":"0000-00-00 00:00:00"},{"id":1764,"id_source":260903,"institutes":["ILC"],"type":"book_chapter","type_order":4,"title":"Semantic annotation of Italian legal texts: a FrameNet-based approach","year":2013,"authors":["Venturi, G."],"authors_source":"Venturi, Giulia","authors_cnr_name":["VENTURI, GIULIA"],"authors_cnr_id":["rp00732"],"authors_cnr_institute":[],"abstract":"The FrameNet approach to text semantic annotation can be a reliable model to make explicit the linguistic information and the semantic content of legal texts. This hypothesis is discussed and empirically demonstrated through an experiment of annotation of a corpus of Italian legal texts. This study is aimed at showing how FrameNet is particularly appropriate in order to provide new perspectives for legal language studies and for legal knowledge representation tasks. Moreover, by relying on the output of an automatic dependency parser, the FrameNet-based annotation methodology presented here is meant to be succesfully used in automatic semantic processing tasks of legal texts","keywords":["Legal Language","Semantic Annotation","Legal Ontologies","Natural Language Processing"],"pages":"51-84","url":"https:\/\/iris.cnr.it\/handle\/20.500.14243\/260903","volume":"","doi":"10.1075\/bct.58","editors":["Fried, M.","Nikiforidou, K."],"editors_source":"Mirjam Fried and Kiki Nikiforidou","published":"Advances in Frame Semantics","publisher":"John Benjamins Publishing Company (Amsterdam\/Philadelphia, USA)","issn":"","isbn":"9789027202772","conference_name":"","conference_place":"Amsterdam\/Philadelphia","conference_date":"","last_updated_cnr":"2024-05-04 11:49:06","last_updated_oai":"2024-05-04 11:49:06","last_updated_www":"0000-00-00 00:00:00"},{"id":544,"id_source":227043,"institutes":["ILC"],"type":"conference_article","type_order":7,"title":"Linguistic Profiling based on General-purpose Features and Native Language Identification","year":2013,"authors":["Cimino, A.","Dell'Orletta, F.","Venturi, G.","Montemagni, S."],"authors_source":"Cimino, Andrea; Dell'Orletta, Felice; Venturi, Giulia; Montemagni, Simonetta","authors_cnr_name":["CIMINO, ANDREA","DELL'ORLETTA, FELICE","VENTURI, GIULIA","MONTEMAGNI, SIMONETTA"],"authors_cnr_id":["rp05770","rp22811","rp00732","rp16780"],"authors_cnr_institute":[],"abstract":"In this paper, we describe our approach to native language identification and discuss the results we submitted as participants to the First NLI Shared Task. By resorting to a wide set of general-purpose features qualifying the lexical and grammatical structure of a text, rather than to ad hoc features specifically selected for the NLI task, we achieved encouraging results, which show that the proposed approach is general-purpose and portable across different tasks, domains and languages","keywords":["Native Language Identification","Linguistic Profiling"],"pages":"207-215","url":"http:\/\/www.aclweb.org\/anthology\/W13-1727","volume":"","doi":"","editors":[],"editors_source":"","published":"","publisher":"","issn":"","isbn":"978-1-937284-47-3","conference_name":"8th workshop on \"Innovative Use of NLP for Building Educational Applications\"","conference_place":"","conference_date":"","last_updated_cnr":"2024-05-19 14:18:24","last_updated_oai":"2024-05-19 14:18:24","last_updated_www":"0000-00-00 00:00:00"},{"id":393,"id_source":227044,"institutes":["ILC"],"type":"conference_article","type_order":7,"title":"Unsupervised Linguistically-Driven Reliable Dependency Parses Detection and Self-Training for Adaptation to the Biomedical Domain","year":2013,"authors":["Dell'Orletta, F.","Venturi, G.","Montemagni, S."],"authors_source":"Felice Dell'Orletta; Giulia Venturi; Simonetta Montemagni","authors_cnr_name":["DELL'ORLETTA, FELICE","VENTURI, GIULIA","MONTEMAGNI, SIMONETTA"],"authors_cnr_id":["rp22811","rp00732","rp16780"],"authors_cnr_institute":[],"abstract":"In this paper, a new self-training method for domain adaptation is illustrated, where the selection of reliable parses is carried out by an unsupervised linguistically-driven algorithm, ULISSE. The method has been tested on biomedical texts with results showing a significant improvement with respect to considered baselines, which demonstrates its ability to capture both reliability of parses and domain-specificity of linguistic constructions","keywords":["Self-training","Domain Adaptation","Biomedical Texts"],"pages":"45-53","url":"http:\/\/www.aclweb.org\/anthology\/W13-1906","volume":"","doi":"","editors":[],"editors_source":"","published":"","publisher":"","issn":"","isbn":"978-1-937284-55-8","conference_name":"12th workshop on \"Biomedical Natural Language Processing\" (BioNLP)","conference_place":"","conference_date":"","last_updated_cnr":"2024-06-05 11:42:41","last_updated_oai":"2024-06-05 11:42:41","last_updated_www":"0000-00-00 00:00:00"},{"id":906,"id_source":304637,"institutes":["ILC"],"type":"conference_article","type_order":7,"title":"Investigating legal language peculiarities across different types of Italian legal texts: an NLP-based approach","year":2013,"authors":["Venturi, G."],"authors_source":"Giulia Venturi","authors_cnr_name":["VENTURI, GIULIA"],"authors_cnr_id":["rp00732"],"authors_cnr_institute":[],"abstract":"In this paper, the author carried out the linguistic profiling of a corpus of different types of Italian legal texts exemplifying different sub-varieties of Italian legal language by relying on a wide range of different linguistic features (lexical, morpho-syntactic and syntactic) automatically extracted from the output of a multi-level automatic linguistic analysis of texts. The devised comparative approach allowed investigating the linguistic variation i) between the considered corpus of legal texts and a corpus of newspaper articles representative of Italian ordinary language and ii) among the considered types of legal texts (legislative acts, administrative acts, the Italian Constitution and legal cases). Achieved results can provide the starting point to identify areas of lexical, morpho-syntactic and\/or syntactic complexity within a legal text in order to assess its readability as well to perform a number of different computational forensic linguistics tasks","keywords":["Legal language analysis","linguistic profiling","legal genres"],"pages":"1-19","url":"http:\/\/ler.letras.up.pt\/uploads\/ficheiros\/13624.pdf","volume":"","doi":"","editors":[],"editors_source":"","published":"","publisher":"","issn":"","isbn":"978-989-8648-14-3","conference_name":"3rd European Conference of the International Association of Forensic Linguists","conference_place":"","conference_date":"","last_updated_cnr":"2024-05-05 19:33:09","last_updated_oai":"2024-05-05 19:33:09","last_updated_www":"0000-00-00 00:00:00"},{"id":2445,"id_source":289319,"institutes":["ILC"],"type":"misc","type_order":11,"title":"Exploring the use of neuroscience in the Italian courtrooms: the linguistic and lexico-semantic analysis of a corpus of Italian case law texts","year":2013,"authors":["Sagri, M. T.","Venturi, G."],"authors_source":"M.T.Sagri; G.Venturi;","authors_cnr_name":["VENTURI, GIULIA"],"authors_cnr_id":["rp00732"],"authors_cnr_institute":[],"abstract":"","keywords":[],"pages":"","url":"https:\/\/iris.cnr.it\/handle\/20.500.14243\/289319","volume":"","doi":"","editors":[],"editors_source":"","published":"","publisher":"","issn":"","isbn":"","conference_name":"","conference_place":"","conference_date":"","last_updated_cnr":"2024-05-17 15:20:03","last_updated_oai":"0000-00-00 00:00:00","last_updated_www":"0000-00-00 00:00:00"},{"id":801,"id_source":289376,"institutes":["ILC"],"type":"conference_article","type_order":7,"title":"Lessico settoriale e lessico comune dell'estrazione di terminologia specialistica da corpora di dominio","year":2012,"authors":["Bonin, F.","Dell'Orletta, F.","Montemagni, S.","Venturi, G."],"authors_source":"Bonin, F; Dell'Orletta, F; Montemagni, S; Venturi, G","authors_cnr_name":["DELL'ORLETTA, FELICE","MONTEMAGNI, SIMONETTA","VENTURI, GIULIA"],"authors_cnr_id":["rp22811","rp16780","rp00732"],"authors_cnr_institute":[],"abstract":"","keywords":[],"pages":"207-220","url":"https:\/\/iris.cnr.it\/handle\/20.500.14243\/289376","volume":"","doi":"","editors":[],"editors_source":"","published":"Lessico e Lessicologia. Atti del XLIV congresso internazionale di studi della societa\u0300 di linguistica italiana","publisher":"Bulzoni Editore (Roma, ITA)","issn":"","isbn":"978-88-7870-655-2","conference_name":"XLIV congresso internazionale di studi della societa\u0300 di linguistica italiana","conference_place":"Roma","conference_date":"","last_updated_cnr":"2024-06-19 07:13:29","last_updated_oai":"2024-06-19 07:13:29","last_updated_www":"0000-00-00 00:00:00"},{"id":1202,"id_source":5141,"institutes":["ILC"],"type":"conference_article","type_order":7,"title":"The SPLeT-2012 Shared Task on Dependency Parsing of Legal Texts","year":2012,"authors":["Dell'Orletta, F.","Marchi, S.","Montemagni, S.","Plank, B.","Venturi, G."],"authors_source":"Dell'Orletta, Felice; Marchi, Simone; Montemagni, Simonetta; Plank, Barbara; Venturi, Giulia","authors_cnr_name":["DELL'ORLETTA, FELICE","MARCHI, SIMONE","MONTEMAGNI, SIMONETTA","VENTURI, GIULIA"],"authors_cnr_id":["rp22811","rp09286","rp16780","rp00732"],"authors_cnr_institute":[],"abstract":"The 4th Workshop on \"Semantic Processing of Legal Texts\" (SPLeT-2012) presents the first multilingual shared task on Dependency Parsing of Legal Texts. In this paper, we define the general task and its internal organization into sub-tasks, describe the datasets and the domain-specific linguistic peculiarities characterizing them. We finally report the results achieved by the participating systems, describe the underlying approaches and provide a first analysis of the final test results","keywords":["Dependency Parsing","Domain Adaptation","Legal Text Processing"],"pages":"","url":"http:\/\/www.lrec-conf.org\/proceedings\/lrec2012\/workshops\/27.LREC%202012%20Workshop%20Proceedings%20SPLeT.pdf","volume":"","doi":"","editors":[],"editors_source":"","published":"","publisher":"","issn":"","isbn":"","conference_name":"Fourth Workshop on Semantic Processing of Legal Texts (SPLeT 2012)-First Shared Task on Dependency Parsing of Legal Texts (SPLeT 2012)","conference_place":"","conference_date":"","last_updated_cnr":"2024-09-19 22:57:57","last_updated_oai":"2024-09-19 22:57:57","last_updated_www":"0000-00-00 00:00:00"},{"id":1417,"id_source":5136,"institutes":["IGSG","ILC"],"type":"conference_article","type_order":7,"title":"Domain Adaptation for Dependency Parsing at Evalita 2011","year":2012,"authors":["Dell'Orletta, F.","Marchi, S.","Montemagni, S.","Venturi, G.","Agnoloni, T.","Francesconi, E."],"authors_source":"Dell'Orletta, Felice; Marchi, Simone; Montemagni, Simonetta; Venturi, Giulia; Agnoloni, Tommaso; Francesconi, Enrico","authors_cnr_name":["DELL'ORLETTA, FELICE","MARCHI, SIMONE","MONTEMAGNI, SIMONETTA","VENTURI, GIULIA","AGNOLONI, TOMMASO","FRANCESCONI, ENRICO"],"authors_cnr_id":["rp22811","rp09286","rp16780","rp00732","rp21506","rp03980"],"authors_cnr_institute":[],"abstract":"The domain adaptation task was aimed at investigating techniques for adapting state-of-the-art dependency parsing systems to new domains. Both the language dealt with, i. e. Italian, and the target domain, namely the legal domain, represent two main novelties of the task organised at Evalita 2011. In this paper, we define the task and describe how the datasets were created from different resources. In addition, we characterize the different approaches of the participating systems, report the test results, and provide a first analysis of these results","keywords":["Dependency Parsing","Domain Adaptation","Legal Text Processing"],"pages":"1-7","url":"http:\/\/www.evalita.it\/sites\/evalita.fbk.eu\/files\/working_notes2011\/Domain_Adaptation\/","volume":"","doi":"","editors":[],"editors_source":"","published":"","publisher":"","issn":"","isbn":"","conference_name":"Evaluation of NLP and Speech Tools for Italian (EVALITA 2011): Domain Adaptation track","conference_place":"","conference_date":"","last_updated_cnr":"2024-09-19 22:59:01","last_updated_oai":"2024-09-19 22:59:01","last_updated_www":"0000-00-00 00:00:00"},{"id":2460,"id_source":266008,"institutes":["ILC"],"type":"conference_article","type_order":7,"title":"Genre-oriented Readability Assessment: a Case Study","year":2012,"authors":["Dell'Orletta, F.","Montemagni, S.","Venturi, G."],"authors_source":"Dell'Orletta F;Montemagni S;VENTURI G.","authors_cnr_name":[],"authors_cnr_id":[],"authors_cnr_institute":[],"abstract":"","keywords":[],"pages":"","url":"https:\/\/iris.cnr.it\/handle\/20.500.14243\/266008","volume":"","doi":"","editors":[],"editors_source":"","published":"","publisher":"","issn":"","isbn":"978-1-62748-389-6","conference_name":"","conference_place":"","conference_date":"","last_updated_cnr":"2024-03-28 19:16:52","last_updated_oai":"0000-00-00 00:00:00","last_updated_www":"0000-00-00 00:00:00"},{"id":2129,"id_source":260805,"institutes":["ILC"],"type":"conference_article","type_order":7,"title":"Enriching the ISST-TANL Corpus with Semantic Frames","year":2012,"authors":["Lenci, A.","Montemagni, S.","Venturi, G.","Cutrulla Maria, R."],"authors_source":"Lenci, Alessandro; Montemagni, Simonetta; Venturi, Giulia; Cutrulla Maria, Rosaria","authors_cnr_name":["MONTEMAGNI, SIMONETTA","VENTURI, GIULIA"],"authors_cnr_id":["rp16780","rp00732"],"authors_cnr_institute":[],"abstract":"The paper describes the design and the results of a manual annotation methodology devoted to enrich the ISST-TANL Corpus with Semantic Frames information. The main issues encountered in applying the English FrameNet annotation criteria to a corpus of Italian language are discussed together with the choice of anchoring the semantic annotation layer to the underlying dependency syntactic structure. We also describe an experiment to measure inter-annotator agreement and a first case study to extend and specialise FrameNet annotation to a corpus of legislative texts","keywords":["Semantic annotation","FrameNet","Multi-layer annotated corpus"],"pages":"3719-3726","url":"http:\/\/www.lrec-conf.org\/proceedings\/lrec2012\/pdf\/986_Paper.pdf","volume":"","doi":"","editors":["Calzolari, N.","Choukri, K.","Declerck, T.","Do\u011fan, M. U.","Maegaard, B.","Mariani, J.","Moreno, A.","Odijk, J.","Piperidis, S."],"editors_source":"Nicoletta Calzolari (Conference Chair) and Khalid Choukri and Thierry Declerck and Mehmet U?ur Do?an and Bente Maegaard and Joseph Mariani and Asuncion Moreno and Jan Odijk and Stelios Piperidis","published":"Proceedings of the Eight International Conference on Language Resources and Evaluation (LREC'12)","publisher":"European language resources association (ELRA) (Paris, FRA)","issn":"","isbn":"978-2-9517408-7-7","conference_name":"Eight International Conference on Language Resources and Evaluation (LREC'12)","conference_place":"Paris","conference_date":"","last_updated_cnr":"2025-03-14 22:17:10","last_updated_oai":"2025-03-14 22:17:10","last_updated_www":"0000-00-00 00:00:00"},{"id":1506,"id_source":175344,"institutes":["ILC"],"type":"journal_article","type_order":1,"title":"The BioLexicon: a large-scale terminological resource for biomedical text mining","year":2011,"authors":["Thompson, P.","McNaught, J.","Montemagni, S.","Calzolari, N.","Del Gratta, R.","Lee, V.","Marchi, S.","Monachini, M.","Pezik, P.","Quochi, V.","Rupp, C.","Sasaki, Y.","Venturi, G.","Rebholzschuhmann, D.","Ananiadou, S."],"authors_source":"Thompson, Paul; Mcnaught, John; Montemagni, Simonetta; Calzolari, Nicoletta; DEL GRATTA, Riccardo; Lee, Vivian; Marchi, Simone; Monachini, Monica; Pezik, Piotr; Quochi, Valeria; Rupp, Cj; Sasaki, Yutaka; Venturi, Giulia; Rebholzschuhmann, Dietrich; Ananiadou, Sophia","authors_cnr_name":["MONTEMAGNI, SIMONETTA","DEL GRATTA, RICCARDO","MARCHI, SIMONE","MONACHINI, MONICA","QUOCHI, VALERIA","VENTURI, GIULIA"],"authors_cnr_id":["rp16780","rp00284","rp09286","rp19457","rp13283","rp00732"],"authors_cnr_institute":[],"abstract":"Background Due to the rapidly expanding body of biomedical literature, biologists require increasingly sophisticated and efficient systems to help them to search for relevant information. Such systems should account for the multiple written variants used to represent biomedical concepts, and allow the user to search for specific pieces of knowledge (or events) involving these concepts, e. g., protein-protein interactions. Such functionality requires access to detailed information about words used in the biomedical literature. Existing databases and ontologies often have a specific focus and are oriented towards human use. Consequently, biological knowledge is dispersed amongst many resources, which often do not attempt to account for the large and frequently changing set of variants that appear in the literature. Additionally, such resources typically do not provide information about how terms relate to each other in texts to describe events. Results This article provides an overview of the design, construction and evaluation of a large-scale lexical and conceptual resource for the biomedical domain, the BioLexicon. The resource can be exploited by text mining tools at several levels, e. g., part-of-speech tagging, recognition of biomedical entities, and the extraction of events in which they are involved. As such, the BioLexicon must account for real usage of words in biomedical texts. In particular, the BioLexicon gathers together different types of terms from several existing data resources into a single, unified repository, and augments them with new term variants automatically extracted from biomedical literature. Extraction of events is facilitated through the inclusion of biologically pertinent verbs (around which events are typically organized) together with information about typical patterns of grammatical and semantic behaviour, which are acquired from domain-specific texts. In order to foster interoperability, the BioLexicon is modelled using the Lexical Markup Framework, an ISO standard. Conclusions The BioLexicon contains over 2. 2 M lexical entries and over 1. 8 M terminological variants, as well as over 3. 3 M semantic relations, including over 2 M synonymy relations. Its exploitation can benefit both application developers and users. We demonstrate some such benefits by describing integration of the resource into a number of different tools, and evaluating improvements in performance that this can bring","keywords":["Text Mining","Information Extraction","Computational Lexicon"],"pages":"1-29","url":"http:\/\/www.biomedcentral.com\/1471-2105\/12\/397","volume":"12 (397)","doi":"10.1186\/1471-2105-12-397","editors":[],"editors_source":"","published":"BMC BIOINFORMATICS","publisher":"","issn":"1471-2105","isbn":"","conference_name":"","conference_place":"","conference_date":"","last_updated_cnr":"2025-03-07 07:10:53","last_updated_oai":"2025-03-07 07:10:53","last_updated_www":"0000-00-00 00:00:00"},{"id":1800,"id_source":279141,"institutes":["ILC"],"type":"journal_article","type_order":1,"title":"Semantic annotation of Italian legal texts: a FrameNet-based approach","year":2011,"authors":["Venturi, G."],"authors_source":"Venturi G.","authors_cnr_name":["VENTURI, GIULIA"],"authors_cnr_id":["rp00732"],"authors_cnr_institute":[],"abstract":"","keywords":[],"pages":"46-79","url":"https:\/\/iris.cnr.it\/handle\/20.500.14243\/279141","volume":"3 (1)","doi":"10.1075\/cf.3.1.02ven","editors":[],"editors_source":"","published":"CONSTRUCTIONS AND FRAMES","publisher":"","issn":"1876-1933","isbn":"","conference_name":"","conference_place":"","conference_date":"","last_updated_cnr":"2024-03-13 20:55:56","last_updated_oai":"2024-03-13 20:55:56","last_updated_www":"0000-00-00 00:00:00"},{"id":1696,"id_source":94003,"institutes":["ILC","IRISS"],"type":"book_chapter","type_order":4,"title":"Tecnologie linguistico-computazionali per il monitoraggio della competenza linguistica italiana degli alunni stranieri nella scuola primaria e secondaria","year":2011,"authors":["Dell'Orletta, F.","Montemagni, S.","Vecchi Eva, M.","Venturi, G."],"authors_source":"Dell'Orletta, Felice; Montemagni, Simonetta; Vecchi Eva, Maria; Venturi, Giulia","authors_cnr_name":["DELL'ORLETTA, FELICE","MONTEMAGNI, SIMONETTA","VENTURI, GIULIA"],"authors_cnr_id":["rp22811","rp16780","rp00732"],"authors_cnr_institute":[],"abstract":"La possibilit\u00e0 di disporre di tecnologie avanzate e innovative che permettano di monitorare la competenza linguistica degli alunni stranieri e, al contempo, valutare l'adeguatezza dei materiali didattici a loro offerti pu\u00f2 essere di supporto all'insegnante nell'orientare la propria azione formativa, rendendo cos\u00ec il processo di integrazione linguistico-culturale meno faticoso e traumatico. In tale ottica, questo studio, realizzato col supporto di una piattaforma ormai consolidata di metodi e strumenti per il trattamento automatico dell'italiano, costituisce il primo tentativo condotto in relazione alla lingua italiana, per mettere a punto una metodologia di monitoraggio linguistico rivolta specificamente agli studenti apprendenti la lingua italiana come L2 ed alle loro produzioni scritte","keywords":["Trattamento Automatico del Linguaggio","Stranieri","Lingua italiana"],"pages":"319-336","url":"https:\/\/iris.cnr.it\/handle\/20.500.14243\/94003","volume":"","doi":"","editors":["Bruno, G. C.","Caruso, I.","Sanna, M.","Vellecco, I."],"editors_source":"Bruno Giovanni Carlo; Caruso Immacolata; Sanna Manuela; Vellecco Immacolata","published":"Percorsi Migranti","publisher":"Mc Graw-Hill (Milano, ITA)","issn":"","isbn":"978-88-386-7296-5","conference_name":"","conference_place":"Milano","conference_date":"","last_updated_cnr":"2024-06-11 12:47:49","last_updated_oai":"2024-06-11 12:47:49","last_updated_www":"0000-00-00 00:00:00"},{"id":84,"id_source":214930,"institutes":["ILC"],"type":"conference_article","type_order":7,"title":"READ-IT: assessing readability of Italian texts with a view to text simplification","year":2011,"authors":["Dell'Orletta, F.","Montemagni, S.","Venturi, G."],"authors_source":"Felice Dell'Orletta; Simonetta Montemagni; Giulia Venturi","authors_cnr_name":["DELL'ORLETTA, FELICE","MONTEMAGNI, SIMONETTA","VENTURI, GIULIA"],"authors_cnr_id":["rp22811","rp16780","rp00732"],"authors_cnr_institute":[],"abstract":"In this paper, we propose a new approach to readability assessment with a specific view to the task of text simplification: the intended audience includes people with low literacy skills and\/or with mild cognitive impairment. READ-IT represents the first advanced readability assessment tool for what concerns Italian, which combines traditional raw text features with lexical, morpho-syntactic and syntactic information. In READ-IT readability assessment is carried out with respect to both documents and sentences where the latter represents an important novelty of the proposed approach creating the prerequisites for aligning the readability assessment step with the text simplification process. READ-IT shows a high accuracy in the document classification task and promising results in the sentence classification scenario","keywords":["Readability Assessment","Text Simplification"],"pages":"73-83","url":"http:\/\/dl.acm.org\/citation.cfm?id=2140511","volume":"","doi":"","editors":[],"editors_source":"","published":"","publisher":"","issn":"","isbn":"978-1-937284-14-5","conference_name":"SLPAT '11 Proceedings of the Second Workshop on Speech and Language Processing for Assistive Technologies","conference_place":"","conference_date":"","last_updated_cnr":"2024-03-31 14:15:12","last_updated_oai":"2024-03-31 14:15:12","last_updated_www":"0000-00-00 00:00:00"},{"id":1194,"id_source":214925,"institutes":["ILC"],"type":"conference_article","type_order":7,"title":"ULISSE: an unsupervised algorithm for detecting reliable dependency parses","year":2011,"authors":["Dell'Orletta, F.","Venturi, G.","Montemagni, S."],"authors_source":"Dell'Orletta, Felice; Venturi, Giulia; Montemagni, Simonetta","authors_cnr_name":["DELL'ORLETTA, FELICE","VENTURI, GIULIA","MONTEMAGNI, SIMONETTA"],"authors_cnr_id":["rp22811","rp00732","rp16780"],"authors_cnr_institute":[],"abstract":"In this paper we present ULISSE, an unsupervised linguistically-driven algorithm to select reliable parses from the output of a dependency parser. Different experiments were devised to show that the algorithm is robust enough to deal with the output of different parsers and with different languages, as well as to be used across different domains. In all cases, ULISSE appears to outperform the baseline algorithms","keywords":["Dependency Parsing","Selection of Reliable Parses","Unsupervised Algorithm"],"pages":"115-124","url":"http:\/\/dl.acm.org\/citation.cfm?id=2018950","volume":"","doi":"","editors":[],"editors_source":"","published":"","publisher":"","issn":"","isbn":"978-1-932432-92-3","conference_name":"CoNLL '11 Proceedings of the Fifteenth Conference on Computational Natural Language Learning","conference_place":"","conference_date":"","last_updated_cnr":"2025-02-09 08:09:33","last_updated_oai":"2025-02-09 08:09:33","last_updated_www":"0000-00-00 00:00:00"},{"id":2507,"id_source":108688,"institutes":["ILC","IRISS"],"type":"misc","type_order":11,"title":"Tecnologie linguistico-computazionali per il monitoraggio della competenza linguistica italiana degli alunni stranieri nella scuola primaria e secondaria","year":2011,"authors":["Venturi, G."],"authors_source":"Venturi, G","authors_cnr_name":["VENTURI, GIULIA"],"authors_cnr_id":["rp00732"],"authors_cnr_institute":[],"abstract":"","keywords":[],"pages":"","url":"https:\/\/iris.cnr.it\/handle\/20.500.14243\/108688","volume":"","doi":"","editors":[],"editors_source":"","published":"","publisher":"","issn":"","isbn":"","conference_name":"Migrazioni \/ Linguaggi, seminario (Roma, 30 marzo 2011)","conference_place":"","conference_date":"","last_updated_cnr":"2024-05-14 19:32:04","last_updated_oai":"2024-05-14 19:32:04","last_updated_www":"0000-00-00 00:00:00"},{"id":1413,"id_source":50352,"institutes":["ILC"],"type":"book_chapter","type_order":4,"title":"Legal Language and Legal Knowledge Management Applications","year":2010,"authors":["Venturi, G."],"authors_source":"Giulia Venturi","authors_cnr_name":["VENTURI, GIULIA"],"authors_cnr_id":["rp00732"],"authors_cnr_institute":[],"abstract":"This work is an investigation into the peculiarities of legal language with respect to ordinary language. Based on the idea that a shallow parsing approach can help to provide enough detailed linguistic information, this work presents the results obtained by shallow parsing (i. e. chunking) corpora of Italian and English legal texts and comparing them with corpora of ordinary language. In particular, this paper puts the emphasis of how understanding the syntactic and lexical characteristics of this specialised language has practical importance in the development of domain-specific Knowledge Management applications","keywords":["Parsing Legal Texts","Natural Language Processing","Legal Language","Knowledge Management Applications"],"pages":"3-26","url":"https:\/\/iris.cnr.it\/handle\/20.500.14243\/50352","volume":"","doi":"","editors":["Francesconi, E.","Montemagni, S.","Peters, W."],"editors_source":"Francesconi E., Montemagni S., Peters W. and Tiscornia D.","published":"Semantic Processing of Legal Texts. Where the Language of Law Meets the Law of Language","publisher":"Springer-Verlag (Berlin Heidelberg, DEU)","issn":"","isbn":"3-642-12836-X","conference_name":"","conference_place":"Berlin Heidelberg","conference_date":"","last_updated_cnr":"2024-05-03 23:38:38","last_updated_oai":"2024-05-03 23:38:38","last_updated_www":"0000-00-00 00:00:00"},{"id":1140,"id_source":65162,"institutes":["ILC"],"type":"conference_article","type_order":7,"title":"A Contrastive Approach to Multi-word Extraction from Domain-specific Corpora","year":2010,"authors":["Bonin, F.","Dell'Orletta, F.","Montemagni, S.","Venturi, G."],"authors_source":"Bonin, F; Dell'Orletta, F; Montemagni, S; Venturi, G","authors_cnr_name":["DELL'ORLETTA, FELICE","MONTEMAGNI, SIMONETTA","VENTURI, GIULIA"],"authors_cnr_id":["rp22811","rp16780","rp00732"],"authors_cnr_institute":[],"abstract":"In this paper we present a novel approach to multi-word terminology extraction combining a well-known automatic term recognition approach, the C-NC value method, with a contrastive ranking technique, aimed at refining obtained results either by filtering noise due to common words or by discerning between semantically different types of terms within heterogeneous terminologies. The proposed methodology has been tested in two case studies carried out in the History of Art and Legal domains with promising results","keywords":["Terminology Extraction","Domain-specific Corpora","Multi-word Expression"],"pages":"3222-3229","url":"https:\/\/iris.cnr.it\/handle\/20.500.14243\/65162","volume":"","doi":"","editors":[],"editors_source":"","published":"","publisher":"","issn":"","isbn":"2-9517408-6-7","conference_name":"Seventh International Conference on Language Resources and Evaluation","conference_place":"","conference_date":"","last_updated_cnr":"2024-04-14 10:43:37","last_updated_oai":"2024-04-14 10:43:37","last_updated_www":"0000-00-00 00:00:00"},{"id":655,"id_source":65168,"institutes":["ILC"],"type":"conference_article","type_order":7,"title":"Contrastive filtering of domain specific multi-word terms from different types of corpora","year":2010,"authors":["Bonin, F.","Dell'Orletta, F.","Venturi, G.","Montemagni, S."],"authors_source":"Bonin, F; Dell'Orletta, F; Venturi, G; Montemagni, S","authors_cnr_name":["DELL'ORLETTA, FELICE","VENTURI, GIULIA","MONTEMAGNI, SIMONETTA"],"authors_cnr_id":["rp22811","rp00732","rp16780"],"authors_cnr_institute":[],"abstract":"In this paper we tackle the challenging task of Multi-word term (MWT) extraction from different types of specialized corpora. Contrastive filtering of previously extracted MWTs results in a considerable increment of acquired domain-specific terms","keywords":["multi-word terms extraction","co"],"pages":"76-79","url":"https:\/\/iris.cnr.it\/handle\/20.500.14243\/65168","volume":"","doi":"","editors":[],"editors_source":"","published":"","publisher":"","issn":"","isbn":"978-7-900268-00-6","conference_name":"The 23rd International Conference on Computational Linguistics (COLING 2010). Multiword Expressions: from Theory to Applications (MWE 2010)","conference_place":"","conference_date":"","last_updated_cnr":"2024-05-28 15:35:56","last_updated_oai":"2024-05-28 15:35:56","last_updated_www":"0000-00-00 00:00:00"},{"id":1979,"id_source":106763,"institutes":["ILC"],"type":"misc","type_order":11,"title":"Lessico settoriale e lessico comune nell'estrazione di terminologia specialistica da corpora di dominio","year":2010,"authors":["Bonin, F.","Dell'Orletta, F.","Montemagni, S.","Venturi, G."],"authors_source":"Bonin F.; Dell'Orletta F.; Montemagni S.; Venturi G.","authors_cnr_name":["DELL'ORLETTA, FELICE","MONTEMAGNI, SIMONETTA","VENTURI, GIULIA"],"authors_cnr_id":["rp22811","rp16780","rp00732"],"authors_cnr_institute":[],"abstract":"","keywords":["Automatic Term Extraction"],"pages":"","url":"https:\/\/iris.cnr.it\/handle\/20.500.14243\/106763","volume":"","doi":"","editors":[],"editors_source":"","published":"","publisher":"","issn":"","isbn":"","conference_name":"XLIV Congresso Internazionale di Studi della Societa\u0300 di Linguistica Italiana","conference_place":"","conference_date":"","last_updated_cnr":"2024-05-26 17:30:34","last_updated_oai":"2024-05-26 17:30:34","last_updated_www":"0000-00-00 00:00:00"},{"id":257,"id_source":155081,"institutes":["ILC","IRISS"],"type":"misc","type_order":11,"title":"Tecnologie linguistico-computazionali per il monitoraggio delle competenze linguistiche di apprendenti l'italiano come L2","year":2010,"authors":["Dell'Orletta, F.","Montemagni, S.","Vecchi, E. M.","Venturi, G."],"authors_source":"Dell'Orletta, F; Montemagni, S; Vecchi, E M; Venturi, G","authors_cnr_name":["MONTEMAGNI, SIMONETTA","VENTURI, GIULIA"],"authors_cnr_id":["rp16780","rp00732"],"authors_cnr_institute":[],"abstract":"","keywords":["Natural Language Processing, Educational Linguistics, Language Learning"],"pages":"","url":"https:\/\/iris.cnr.it\/handle\/20.500.14243\/155081","volume":"","doi":"","editors":[],"editors_source":"","published":"","publisher":"","issn":"","isbn":"","conference_name":"Congresso \"IT. L2: italiano lingua seconda nell'universita\u0300, nella scuola e sul territorio. Esperienze didattiche e ricerche\" Universita\u0300 del Piemonte Orientale \"Amedeo Avogadro\", Facolta\u0300 di Lettere e Filosofia","conference_place":"","conference_date":"","last_updated_cnr":"2024-05-22 11:49:45","last_updated_oai":"2024-05-22 11:49:45","last_updated_www":"0000-00-00 00:00:00"},{"id":1965,"id_source":165690,"institutes":["ILC","ITTIG","IGSG"],"type":"book_chapter","type_order":4,"title":"A two-level Knowledge approach to support multilingual legislative drafting","year":2009,"authors":["Agnoloni, T.","Bacci, L.","Francesconi, E.","Peters, W.","Montemagni, S.","Venturi, G."],"authors_source":"Agnoloni, T; Bacci, L; Francesconi, E; Peters, W; Montemagni, S; Venturi, G","authors_cnr_name":["AGNOLONI, TOMMASO","BACCI, LORENZO","FRANCESCONI, ENRICO","MONTEMAGNI, SIMONETTA","VENTURI, GIULIA"],"authors_cnr_id":["rp21506","rp24033","rp03980","rp16780","rp00732"],"authors_cnr_institute":[],"abstract":"","keywords":["DALOS project","Ontological-linguistic"],"pages":"","url":"https:\/\/iris.cnr.it\/handle\/20.500.14243\/165690","volume":"","doi":"","editors":[],"editors_source":"","published":"","publisher":"","issn":"","isbn":"","conference_name":"","conference_place":"","conference_date":"","last_updated_cnr":"2024-05-07 11:02:11","last_updated_oai":"2024-05-07 11:02:11","last_updated_www":"0000-00-00 00:00:00"},{"id":1797,"id_source":134815,"institutes":["ILC"],"type":"book_chapter","type_order":4,"title":"Ontology learning from Italian legal texts","year":2009,"authors":["Lenci, A.","Montemagni, S.","Pirrelli, V.","Venturi, G."],"authors_source":"Lenci A.; Montemagni S.; Pirrelli V.; Giulia V.","authors_cnr_name":["MONTEMAGNI, SIMONETTA","PIRRELLI, VITO"],"authors_cnr_id":["rp16780","rp03420"],"authors_cnr_institute":[],"abstract":"The paper reports on the methodology and preliminary results of a case study in automatically extracting ontological knowledge from Italian legislative texts. We use a fully-implemented ontology learning system (T2K) that includes a battery of tools for Natural Language Processing (NLP), statistical text analysis and machine language learning. Tools are dynamically integrated to provide an incremental representation of the content of vast repositories of unstructured documents. Evaluated results, however preliminary, show the great potential of NLP-powered incremental systems like T2K for accurate large-scale semi-automatic extraction of legal ontologies","keywords":["Ontology Learning","document management","legal knowledge extraction"],"pages":"75-94","url":"https:\/\/iris.cnr.it\/handle\/20.500.14243\/134815","volume":"","doi":"10.3233\/978-1-58603-942-4-75","editors":["Breuker, J.","Casanovas, P.","Klein, M. C. A.","Francesconi, E."],"editors_source":"Joost Breuker; Pompeu Casanovas; Michel C.A. Klein; Enrico Francesconi","published":"Law, Ontologies and the Semantic Web-Channelling the Legal Information Flood","publisher":"","issn":"","isbn":"978-1-58603-942-4","conference_name":"","conference_place":"","conference_date":"","last_updated_cnr":"2024-03-04 01:05:08","last_updated_oai":"2024-03-04 01:05:08","last_updated_www":"0000-00-00 00:00:00"},{"id":2560,"id_source":1330,"institutes":["IGSG","ILC","ITTIG"],"type":"conference_article","type_order":7,"title":"NLP based Metadata Extraction for Legal Text Consolidation","year":2009,"authors":["Spinosa, P.","Giardiello, G.","Cherubini, M.","Marchi, S.","Venturi, G.","Montemagni, S."],"authors_source":"Spinosa P.; Giardiello G.; Cherubini M.; Marchi S.; Venturi G.; Montemagni S.","authors_cnr_name":["SPINOSA, PIERLUIGI","GIARDIELLO, GERARDO","CHERUBINI, MANOLA","MARCHI, SIMONE","VENTURI, GIULIA","MONTEMAGNI, SIMONETTA"],"authors_cnr_id":["rp21742","rp25089","rp00903","rp09286","rp00732","rp16780"],"authors_cnr_institute":[],"abstract":"The paper describes a system for the automatic consolidation of Italian legislative texts to be used as a support of an editorial consolidating activity and dealing with the following typology of textual amendments: repeal, substitution and integration. The focus of the paper is on the semantic analysis of the textual amendment provisions and the formalized representation of the amendments in terms of metadata. The proposed approach to consolidation is metadata-oriented and based on Natural Language Processing (NLP) techniques: we use XML-based standards for metadata annotation of legislative acts and a flexible NLP architecture for extracting metadata from parsed texts. An evaluation of achieved results is also provided","keywords":["Natural Language Processing","textual amendments","XML representation","metadata extraction","consolidation of legal text"],"pages":"40-49","url":"https:\/\/iris.cnr.it\/handle\/20.500.14243\/1330","volume":"","doi":"10.1145\/1568234.1568240","editors":["Hafner, C. D.","Casanovas, P."],"editors_source":"Carole D. Hafner; Pompeu Casanovas","published":"Proceeding ICAIL '09 Proceedings of the 12th International Conference on Artificial Intelligence and Law","publisher":"Association Of Computing Machinery (ACM) (New York, USA)","issn":"","isbn":"978-1-60558-597-0","conference_name":"Twelfth International Conference on Artificial Intelligence and Law (ICAIL 2009)","conference_place":"New York","conference_date":"","last_updated_cnr":"2024-03-31 12:51:21","last_updated_oai":"2024-03-31 12:51:21","last_updated_www":"0000-00-00 00:00:00"},{"id":1231,"id_source":155070,"institutes":["ILC","ITTIG","IGSG"],"type":"conference_article","type_order":7,"title":"Towards a FrameNet Resource for the Legal Domain","year":2009,"authors":["Venturi, G.","Lenci, A.","Montemagni, S.","Vecchi, E. M.","Sagri, M. T.","Tiscornia, D.","Agnoloni, T."],"authors_source":"Venturi, G; Lenci, A; Montemagni, S; Vecchi, E M; Sagri, M T; Tiscornia, D; Agnoloni, T","authors_cnr_name":["VENTURI, GIULIA","MONTEMAGNI, SIMONETTA","SAGRI, MARIA TERESA","TISCORNIA, DANIELA","AGNOLONI, TOMMASO"],"authors_cnr_id":["rp00732","rp16780","rp17388","rp22053","rp21506"],"authors_cnr_institute":[],"abstract":"","keywords":["Frame Semantics","Legal Ontologies","Knowledge Representation","Corpus Annotation"],"pages":"","url":"https:\/\/iris.cnr.it\/handle\/20.500.14243\/155070","volume":"","doi":"","editors":[],"editors_source":"","published":"","publisher":"","issn":"","isbn":"","conference_name":"3rd Workshop on Legal Ontologies and Artificial Intelligence Techniques joint with 2nd Workshop on Semantic Processing of Legal text","conference_place":"","conference_date":"","last_updated_cnr":"2024-06-04 21:55:47","last_updated_oai":"2024-06-04 21:55:47","last_updated_www":"0000-00-00 00:00:00"},{"id":1886,"id_source":65110,"institutes":["ILC"],"type":"conference_article","type_order":7,"title":"Bootstrapping a Verb Lexicon for Biomedical Information Extraction","year":2009,"authors":["Venturi, G.","Montemagni, S.","Marchi, S.","Sasaki, Y.","Thompson, P.","McNaught, J.","Ananiadou, S."],"authors_source":"Venturi G.; Montemagni S.; Marchi S.; Sasaki Y.; Thompson P.; McNaught J.; Ananiadou S.","authors_cnr_name":["VENTURI, GIULIA","MONTEMAGNI, SIMONETTA","MARCHI, SIMONE"],"authors_cnr_id":["rp00732","rp16780","rp09286"],"authors_cnr_institute":[],"abstract":"The extraction of information from texts requires resources that contain both syntactic and semantic properties of lexical units. As the use of language in specialized domains, such as biology, can be very different to the general domain, there is a need for domain-specific resources to ensure that the information extracted is as accurate as possible. We are building a large-scale lexical resource for the biology domain, providing information about predicate-argument structure that has been bootstrapped from a biomedical corpus on the subject of E. Coli. The lexicon is currently focussed on verbs, and includes both automatically-extracted syntactic subcategorization frames, as well as semantic event frames that are based on annotation by domain experts. In addition, the lexicon contains manually-added explicit links between semantic and syntactic slots in corresponding frames. To our knowledge, this lexicon currently represents a unique resource within in the biomedical domain","keywords":["domain-specific lexical resources","Biological Language Processing","syntax-semantic linking"],"pages":"137-148","url":"https:\/\/iris.cnr.it\/handle\/20.500.14243\/65110","volume":"","doi":"10.1007\/978-3-642-00382-0_11","editors":[],"editors_source":"","published":"","publisher":"Springer-Verlag (Berlin Heidelberg, DEU)","issn":"","isbn":"9783642003813","conference_name":"10th International Conference on Intelligent Text Processing and Computational Linguistics","conference_place":"Berlin Heidelberg","conference_date":"","last_updated_cnr":"2024-06-25 17:24:58","last_updated_oai":"2024-06-25 17:24:58","last_updated_www":"0000-00-00 00:00:00"},{"id":752,"id_source":90828,"institutes":["ILC","ITTIG","IGSG"],"type":"misc","type_order":11,"title":"NLP based Metadata Extraction for Legal Text Consolidation","year":2009,"authors":["Spinosa, P.","Giardiello, G.","Cherubini, M.","Marchi, S.","Venturi, G.","Montemagni, S."],"authors_source":"Spinosa P.; Giardiello G.; Cherubini M.; Marchi S.; Venturi G.; Montemagni S.","authors_cnr_name":["SPINOSA, PIERLUIGI","GIARDIELLO, GERARDO","CHERUBINI, MANOLA","MARCHI, SIMONE","VENTURI, GIULIA","MONTEMAGNI, SIMONETTA"],"authors_cnr_id":["rp21742","rp25089","rp00903","rp09286","rp00732","rp16780"],"authors_cnr_institute":[],"abstract":"","keywords":["Natural Language Processing","textual amendments","XML representation","metadata extraction","consolidation of legal text"],"pages":"","url":"https:\/\/iris.cnr.it\/handle\/20.500.14243\/90828","volume":"","doi":"","editors":[],"editors_source":"","published":"","publisher":"","issn":"","isbn":"","conference_name":"Twelfth International Conference on Artificial Intelligence and Law (ICAIL 2009)","conference_place":"","conference_date":"","last_updated_cnr":"2024-05-07 11:13:44","last_updated_oai":"2024-05-07 11:13:44","last_updated_www":"0000-00-00 00:00:00"},{"id":814,"id_source":106756,"institutes":["ILC"],"type":"misc","type_order":11,"title":"Bootstrapping a Verb Lexicon for Biomedical Information Extraction","year":2009,"authors":["Venturi, G.","Montemagni, S.","Marchi, S.","Sasaki, Y.","Thompson, P.","McNaught, J.","Ananiadou, S."],"authors_source":"Venturi, Giulia; Montemagni, Simonetta; Marchi, Simone; Sasaki, Yutaka; Thompson, Paul; Mcnaught, John; Ananiadou, Sophia","authors_cnr_name":["VENTURI, GIULIA","MONTEMAGNI, SIMONETTA","MARCHI, SIMONE"],"authors_cnr_id":["rp00732","rp16780","rp09286"],"authors_cnr_institute":[],"abstract":"The extraction of information from texts requires resources that contain both syntactic and semantic properties of lexical units. As the use Of language in specialized domains, such as biology, can be very different to the general domain, there is a need for domain-specific resources to ensure that the information extracted is as accurate as possible. We are building a large-scale lexical resource for the biology domain. providing information about predicate-argument structure that has been bootstrapped from a biomedical corpus on the subject of E. Coli. The lexicon is currently focussed on verbs, and includes both automatically-extracted syntactic subcategorization frames, as well as semantic event frames that are based on annotation by domain experts. In addition, the lexicon contains manually-added explicit links between semantic and syntactic slots in corresponding frames. To Our knowledge, this lexicon currently represents a unique resource within in the biomedical domain","keywords":["domain-specific lexical resources","lexical acquisition","syntax-semantics linking","Information Extraction","Biological Language Processing"],"pages":"137-148","url":"https:\/\/iris.cnr.it\/handle\/20.500.14243\/106756","volume":"5449","doi":"","editors":["Gelbukh, A."],"editors_source":"Alexander Gelbukh","published":"","publisher":"","issn":"","isbn":"978-3-642-00381-3","conference_name":"International Conference on Intelligent Text Processing and Computational Linguistics (CICLing 2009)","conference_place":"","conference_date":"","last_updated_cnr":"2025-04-05 23:57:20","last_updated_oai":"2025-04-05 23:57:20","last_updated_www":"0000-00-00 00:00:00"},{"id":251,"id_source":37713,"institutes":["ILC"],"type":"journal_article","type_order":1,"title":"Dal testo alla conoscenza e ritorno: estrazione terminologica e annotazione semantica di basi documentali di dominio","year":2008,"authors":["Dell'Orletta, F.","Lenci, A.","Marchi, S.","Montemagni, S.","Pirrelli, V.","Venturi, G."],"authors_source":"Dell'Orletta F.; Lenci A.; Marchi S.; Montemagni S.; Pirrelli V.; Venturi G.","authors_cnr_name":["DELL'ORLETTA, FELICE","MARCHI, SIMONE","MONTEMAGNI, SIMONETTA","PIRRELLI, VITO","VENTURI, GIULIA"],"authors_cnr_id":["rp22811","rp09286","rp16780","rp03420","rp00732"],"authors_cnr_institute":[],"abstract":"The paper focuses on the automatic extraction of domain knowledge from Italian legal texts and presents a fully-implemented ontology learning system (T2K, Text-2-Knowledge) that includes a battery of tools for Natural Language Processing, statistical text analysis and machine learning. Evaluated results show the considerable potential of systems like T2K, exploiting an incremental interleaving of NLP and machine learning techniques for accurate large-scale semi-automatic extraction and structuring of domain-specific knowledge","keywords":["Natural Language Processing","Machine Learning","Knowledge extraction from texts","Ontology learning","Legal ontologies"],"pages":"197-218","url":"https:\/\/iris.cnr.it\/handle\/20.500.14243\/37713","volume":"26 (1-2)","doi":"","editors":[],"editors_source":"","published":"AIDA INFORMAZIONI (ONLINE)","publisher":"","issn":"1594-2201","isbn":"","conference_name":"","conference_place":"","conference_date":"","last_updated_cnr":"2024-10-10 22:49:21","last_updated_oai":"2024-10-10 22:49:21","last_updated_www":"0000-00-00 00:00:00"},{"id":243,"id_source":65083,"institutes":["ILC"],"type":"conference_article","type_order":7,"title":"Dal testo alla conoscenza e ritorno: estrazione terminologica e annotazione semantica di basi documentali di dominio","year":2008,"authors":["Dell'Orletta, F.","Lenci, A.","Marchi, S.","Montemagni, S.","Pirrelli, V.","Venturi, G."],"authors_source":"Dell'Orletta, Felice; Lenci, Alessando; Marchi, Simone; Montemagni, Simonetta; Pirrelli, Vito; Venturi, Giulia","authors_cnr_name":["DELL'ORLETTA, FELICE","MARCHI, SIMONE","MONTEMAGNI, SIMONETTA","PIRRELLI, VITO","VENTURI, GIULIA"],"authors_cnr_id":["rp22811","rp09286","rp16780","rp03420","rp00732"],"authors_cnr_institute":[],"abstract":"The paper focuses on the automatic extraction of domain knowledge from Italian legal texts and presents a fully-implemented ontology learning system (T2K, Text-2-Knowledge) that includes a battery of tools for Natural Language Processing, statistical text analysis and machine learning. Evaluated results show the considerable potential of systems like T2K, exploiting an incremental interleaving of NLP and machine learning techniques for accurate large-scale semi-automatic extraction and structuring of domain-specific knowledge","keywords":["Natural Language Processing","Machine Learning","Knowledge extraction from texts","Ontology learning","Legal ontologies"],"pages":"197-218","url":"http:\/\/www.assiterm91.it\/wp-content\/uploads\/2010\/11\/Convegno-2008.pdf","volume":"ANNO 26, NUMERO 1-2","doi":"","editors":[],"editors_source":"","published":"AIDA INFORMAZIONI","publisher":"","issn":"1121-0095","isbn":"","conference_name":"Atti del Convegno Nazionale Ass. I. Term","conference_place":"","conference_date":"","last_updated_cnr":"2024-04-06 10:53:15","last_updated_oai":"2024-04-06 10:53:15","last_updated_www":"0000-00-00 00:00:00"},{"id":1389,"id_source":65074,"institutes":["ILC"],"type":"conference_article","type_order":7,"title":"Acquiring Legal Ontologies from Domain-specific Texts","year":2008,"authors":["Dell'Orletta, F.","Lenci, A.","Montemagni, S.","Marchi, S.","Pirrelli, V.","Venturi, G."],"authors_source":"Dell'Orletta, F; Lenci, A; Montemagni, S; Marchi, S; Pirrelli, V; Venturi, G","authors_cnr_name":["DELL'ORLETTA, FELICE","MONTEMAGNI, SIMONETTA","MARCHI, SIMONE","PIRRELLI, VITO","VENTURI, GIULIA"],"authors_cnr_id":["rp22811","rp16780","rp09286","rp03420","rp00732"],"authors_cnr_institute":[],"abstract":"The paper reports on methodology and preliminary results ofa case study in automatically extracting ontological knowledgefrom Italian legislative texts in the environmental domain. Weuse a fully-implemented ontology learning system (T2K) thatincludes a battery of tools for Natural Language Processing(NLP), statistical text analysis and machine language learn-ing. Tools are dynamically integrated to provide an incremen-tal representation of the content of vast repositories of unstruc-tured documents. Evaluated results, however preliminary, arevery encouraging, showing the great potential of NLP-poweredincremental systems like T2K for accurate large-scale semi-automatic extraction of legal ontologies","keywords":["Ontology learning","Document management","knowledge extraction from texts","Natural Language Processing"],"pages":"98-101","url":"https:\/\/iris.cnr.it\/handle\/20.500.14243\/65074","volume":"","doi":"","editors":[],"editors_source":"","published":"","publisher":"","issn":"","isbn":"","conference_name":"LangTech 2008","conference_place":"","conference_date":"","last_updated_cnr":"2024-04-19 23:17:04","last_updated_oai":"2024-04-19 23:17:04","last_updated_www":"0000-00-00 00:00:00"},{"id":1199,"id_source":65080,"institutes":["ILC"],"type":"conference_article","type_order":7,"title":"Building a Bio-Event Annotated Corpus for the Acquisition of Semantic Frames from Biomedical Corpora","year":2008,"authors":["Thompson, P.","Cotter, P.","Ananiadou, S.","McNaught, J.","Montemagni, S.","Trabucco, A.","Venturi, G."],"authors_source":"Thompson P.; Cotter P.; Ananiadou S.; McNaught J.; Montemagni S.; Trabucco A.; Venturi G.","authors_cnr_name":["MONTEMAGNI, SIMONETTA","VENTURI, GIULIA"],"authors_cnr_id":["rp16780","rp00732"],"authors_cnr_institute":[],"abstract":"","keywords":["Corpus (creation","annotation","etc.)","Text mining","Semantics","Event Extraction"],"pages":"2159-2166","url":"https:\/\/iris.cnr.it\/handle\/20.500.14243\/65080","volume":"","doi":"","editors":[],"editors_source":"","published":"","publisher":"","issn":"","isbn":"2-9517408-4-0","conference_name":"LREC 2008, Sixth International Conference on Language Resouces and Evaluation","conference_place":"","conference_date":"","last_updated_cnr":"2024-05-17 16:51:54","last_updated_oai":"2024-05-17 16:51:54","last_updated_www":"0000-00-00 00:00:00"},{"id":349,"id_source":65081,"institutes":["ILC"],"type":"conference_article","type_order":7,"title":"Categorising Modality in Biomedical Texts","year":2008,"authors":["Thompson, P.","Venturi, G.","McNaught, J.","Montemagni, S.","Ananiadou, S."],"authors_source":"Thompson, P; Venturi, G; Mcnaught, J; Montemagni, S; Ananiadou, S","authors_cnr_name":["VENTURI, GIULIA","MONTEMAGNI, SIMONETTA"],"authors_cnr_id":["rp00732","rp16780"],"authors_cnr_institute":[],"abstract":"The accurate recognition of modal information is vital for the correct interpretation of statements. In this paper, we report on the collection a list of words and phrases that express modal information in biomedical texts, and propose a categorisation scheme according to the type of information conveyed. We have performed a small pilot study through the annotation of 202 MEDLINE abstracts according to our proposed scheme. Our initial results suggest that modality in biomedical statements can be predicted fairly reliably though the presence of particular lexical items, together with a small amount of contextual information","keywords":["Biomedical texts","Modality"],"pages":"27-34","url":"https:\/\/iris.cnr.it\/handle\/20.500.14243\/65081","volume":"","doi":"","editors":[],"editors_source":"","published":"","publisher":"","issn":"","isbn":"2-9517408-4-0","conference_name":"LREC 2008, Sixth International Conference on Language Resources and Evaluation: Workshop 'Building and Evaluating Resources for Biomedical Text Mining'","conference_place":"","conference_date":"","last_updated_cnr":"2024-05-31 11:29:51","last_updated_oai":"2024-05-31 11:29:51","last_updated_www":"0000-00-00 00:00:00"},{"id":866,"id_source":143465,"institutes":["ILC","ITTIG","IGSG"],"type":"conference_article","type_order":7,"title":"Building an ontological support for multilingual legislative drafting","year":2007,"authors":["Agnoloni, T.","Bacci, L.","Francesconi, E.","Spinosa, P.","Tiscornia, D.","Montemagni, S.","Venturi, G."],"authors_source":"Agnoloni, T; Bacci, L; Francesconi, E; Spinosa, P; Tiscornia, D; Montemagni, S; Venturi, G","authors_cnr_name":["AGNOLONI, TOMMASO","BACCI, LORENZO","FRANCESCONI, ENRICO","SPINOSA, PIERLUIGI","TISCORNIA, DANIELA","MONTEMAGNI, SIMONETTA","VENTURI, GIULIA"],"authors_cnr_id":["rp21506","rp24033","rp03980","rp21742","rp22053","rp16780","rp00732"],"authors_cnr_institute":[],"abstract":"","keywords":[],"pages":"9-18","url":"https:\/\/iris.cnr.it\/handle\/20.500.14243\/143465","volume":"","doi":"","editors":["Ar, L.","Mommers, L."],"editors_source":"Lodder Ar; Mommers L.","published":"Legal Knowledge and information Systems","publisher":"","issn":"","isbn":"","conference_name":"International Conference on Legal Knowledge and Information Systems (JURIX 2007)","conference_place":"","conference_date":"","last_updated_cnr":"2024-05-01 01:28:54","last_updated_oai":"2024-05-01 01:28:54","last_updated_www":"0000-00-00 00:00:00"},{"id":1407,"id_source":65070,"institutes":["ILC"],"type":"conference_article","type_order":7,"title":"NLP-based ontology learning from legal texts. A case study","year":2007,"authors":["Lenci, A.","Montemagni, S.","Pirrelli, V.","Venturi, G."],"authors_source":"Lenci, A; Montemagni, S; Pirrelli, V; Venturi, G","authors_cnr_name":["MONTEMAGNI, SIMONETTA","PIRRELLI, VITO","VENTURI, GIULIA"],"authors_cnr_id":["rp16780","rp03420","rp00732"],"authors_cnr_institute":[],"abstract":"The paper reports on the methodology and preliminary results of a case study in automatically extracting ontological knowledge from Italian legislative texts in the environmental domain. We use a fully-implemented ontology learning system (T2K) that includes a battery of tools for Natural Language Processing (NLP), statistical text analysis and machine language learning. Tools are dynamically integrated to provide an incremental representation of the content of vast repositories of unstructured documents. Evaluated results, however preliminary, are very encouraging, showing the great potential of NLP-powered incremental systems like T2K for accurate large-scale semi-automatic extraction of legal ontologies","keywords":[],"pages":"113-129","url":"https:\/\/iris.cnr.it\/handle\/20.500.14243\/65070","volume":"","doi":"","editors":[],"editors_source":"","published":"","publisher":"","issn":"","isbn":"","conference_name":"II Workshop on Legal Ontologies and Artificial Intelligence Techniques (LOAIT'07)","conference_place":"","conference_date":"","last_updated_cnr":"2024-06-05 12:41:38","last_updated_oai":"2024-06-05 12:41:38","last_updated_www":"0000-00-00 00:00:00"},{"id":1927,"id_source":195951,"institutes":["ILC"],"type":"technical_report","type_order":9,"title":"Report on Ontology learning tool and testing","year":2007,"authors":["Montemagni, S.","Marchi, S.","Venturi, G.","Bartolini, R.","Bertagna, F.","Ruffolo, P.","Peters, W.","Tiscornia, D."],"authors_source":"Montemagni S.; Marchi S.; Venturi G.; Bartolini R.; Bertagna F.; Ruffolo P.; Peters W.; Tiscornia D.","authors_cnr_name":["MONTEMAGNI, SIMONETTA","MARCHI, SIMONE","VENTURI, GIULIA","BARTOLINI, ROBERTO"],"authors_cnr_id":["rp16780","rp09286","rp00732","rp00239"],"authors_cnr_institute":[],"abstract":"This deliverable documents the work done within the DALOS EU project for what concerns the definition and implementation of methodologies and techniques to bootstrap terminological and ontological knowledge from domain corpora. Starting from a corpus of legacy legislative texts in different languages, linguistic technologies combined with statistical techniques have been used to extract significant terms as well as to structure them in conceptual structures for the different languages dealt with within the project, namely Italian, English, Spanish and Dutch","keywords":["Ontology Learning","Term Extraction","Natural Language Processing","Conceptual Indexing"],"pages":"","url":"https:\/\/iris.cnr.it\/handle\/20.500.14243\/195951","volume":"","doi":"","editors":[],"editors_source":"","published":"","publisher":"","issn":"","isbn":"","conference_name":"","conference_place":"","conference_date":"","last_updated_cnr":"2024-03-28 19:10:21","last_updated_oai":"2024-03-28 19:10:21","last_updated_www":"0000-00-00 00:00:00"},{"id":2031,"id_source":195937,"institutes":["ILC"],"type":"technical_report","type_order":9,"title":"Bio-Event Linguistic Annotation Tool. User Manual","year":2007,"authors":["Montemagni, S.","Trabucco, A.","Venturi, G."],"authors_source":"Montemagni S.; Trabucco A.; Venturi G.","authors_cnr_name":["MONTEMAGNI, SIMONETTA","VENTURI, GIULIA"],"authors_cnr_id":["rp16780","rp00732"],"authors_cnr_institute":[],"abstract":"","keywords":[],"pages":"","url":"https:\/\/iris.cnr.it\/handle\/20.500.14243\/195937","volume":"","doi":"","editors":[],"editors_source":"","published":"","publisher":"","issn":"","isbn":"","conference_name":"","conference_place":"","conference_date":"","last_updated_cnr":"2024-05-17 15:38:35","last_updated_oai":"2024-05-17 15:38:35","last_updated_www":"0000-00-00 00:00:00"},{"id":1522,"id_source":457839,"institutes":["ILC"],"type":"technical_report","type_order":9,"title":"Event annotation of domain corpora","year":2007,"authors":["Montemagni, S.","Trabucco, A.","Venturi, G.","Thompson, P.","Cotter, P.","Ananiadou, S.","McNaught, J.","Kim, J.","Rebholz, D.","Pezik, P."],"authors_source":"Montemagni, S; Trabucco, A; Venturi, G; Thompson, P; Cotter, P; Ananiadou, S; Mcnaught, J; Kim, J; Rebholz, D; Pezik, P","authors_cnr_name":["MONTEMAGNI, SIMONETTA","VENTURI, GIULIA"],"authors_cnr_id":["rp16780","rp00732"],"authors_cnr_institute":[],"abstract":"","keywords":[],"pages":"","url":"https:\/\/iris.cnr.it\/handle\/20.500.14243\/457839","volume":"","doi":"","editors":[],"editors_source":"","published":"","publisher":"","issn":"","isbn":"","conference_name":"","conference_place":"","conference_date":"","last_updated_cnr":"2024-04-17 00:39:25","last_updated_oai":"2024-04-17 00:39:25","last_updated_www":"0000-00-00 00:00:00"}]