[{"id":458497,"id_source":596303,"institutes":["ILC"],"type":"conference_article","type_order":7,"title":"FAIR Training Materials for Disciplinary Research Infrastructures: Metadata, Vocabularies, and Selective Harvesting in the H2IOSC Ecosystem","year":2026,"authors":["Melaccio, D.","Pedonese, G.","Frontini, F.","Van Der Lek, I.","Kontino, T."],"authors_source":"Daniele Melaccio, Giulia Pedonese, Francesca Frontini, Iulianna van der Lek, Thalassia Kontino","authors_cnr_name":[],"authors_cnr_id":[],"authors_cnr_institute":[],"abstract":"In a lifelong learning society and especially now in the era of Artificial Intelligence, high-quality, flexible learning resources are essential for acquiring the new skills required by the ever-evolving job market. Research infrastructures play a central role in providing users with specialised training and advanced digital tools, fostering innovation in their fields. CLARIN ERIC, the Common Language Resources and Technology Infrastructure, has consistently promoted training initiatives in linguistic and language technologies, especially through the CLARIN Learning Hub. The Italian national consortium CLARIN-IT has developed this approach within the NRRP project H2IOSC, aiming to create a cluster of research infrastructures in the field of Social Sciences and Humanities. In this context, CLARIN-IT led the work package dedicated to training and developed two platforms: an e-learning portal and a digital library for long-term deposit and curation of training materials as FAIR digital objects. The Skills4EOSC FAIR-by-Design methodology and the SSHOC vocabularies adopted in the H2IOSC project ensured metadata compatibility and interoperability of learning resources across domains. Moreover, the adopted standards will enable harvesting resources in the H2IOSC Marketplace and selective harvesting for domain-specific platforms such as the CLARIN Learning Resource Catalogue. This paper proposes an architecture for metadata mapping and harvesting of training resources to maximise their reusability across infrastructures","keywords":["FAIR learning resources, metadata interoperability, CLARIN, H2IOSC, SSHOC, selective harvesting"],"pages":"","url":"","volume":"","doi":"","editors":[],"editors_source":"","published":"Proceedings of the 22nd Conference on Information and Research Science Connecting to Digital and Library Science","publisher":"","issn":"","isbn":"","conference_name":"Information and Research Science Connecting to Digital and Library Science 2026","conference_place":"","conference_date":"","last_updated_cnr":"0000-00-00 00:00:00","last_updated_oai":"2026-08-29 01:16:00","last_updated_www":"0000-00-00 00:00:00"},{"id":583,"id_source":579261,"institutes":["ILC"],"type":"journal_article","type_order":1,"title":"La repre\u0301sentation et la diffusion des donne\u0301es terminologiques plurilingues: la collection REALITER \u2013 OTPL (CLARIN-IT)","year":2025,"authors":["Dankova, K.","Frontini, F.","Khan, A. F.","Monachini, M."],"authors_source":"Klara Dankova, Francesca Frontini, Anas Fahad Khan, Monica Monachini","authors_cnr_name":[],"authors_cnr_id":[],"authors_cnr_institute":[],"abstract":"In today\u2019s increasingly interconnected world, plurilingual and pluricultural competences are essential for participation in economic, scientific, and cultural exchanges. In this context, the creation and dissemination of plurilingual terminological resources play an important role in ensuring clear and effective communication in scientific and professional fields. The Pan-Latin Terminology Network REALITER recognizes the benefits of plurilingual communication in specialized domains and therefore carries out activities aimed at promoting linguistic diversity in the area of Romance languages. Since its creation (1993), several plurilingual lexicons covering a wide range of sectors, such as the environment, digital technologies, education, and fashion, have been produced. Thanks to the collaboration between CLARIN-IT (the Italian national node of CLARIN ERIC, the European infrastructure for language resources and technologies) and OTPL (Osservatorio di Terminologie e Politiche Linguistiche, Universita\u0300 Cattolica del Sacro Cuore, Milan), these terminological data are indexed in the REALITER \u2013 OTPL collection and published on the ILC4CLARIN-SKOSMOS Service platform. After presenting the REALITER projects, with particular attention to terminological variation and cultural aspects, the paper aims to describe the methodological choices made for the representation and dissemination of these plurilingual lexicons in compliance with FAIR principles. This will highlight the crucial role of infrastructures such as CLARIN ERIC in supporting the representation, sharing, and preservation of this rich linguistic and cultural heritage","keywords":["plurilingualism, lexicon, terminological variation, FAIR principles, infrastructure","plurilinguisme, lexique, variation terminologique, principes FAIR, infrastructure"],"pages":"209-236","url":"https:\/\/id.erudit.org\/iderudit\/1124416ar","volume":"XXXVIII (2)","doi":"10.7202\/1124416ar","editors":[],"editors_source":"","published":"TTR","publisher":"","issn":"","isbn":"","conference_name":"","conference_place":"","conference_date":"","last_updated_cnr":"0000-00-00 00:00:00","last_updated_oai":"2026-05-05 01:13:33","last_updated_www":"0000-00-00 00:00:00"},{"id":1319,"id_source":552963,"institutes":["ILC","OVI","ISPF","ILIESI","INO"],"type":"journal_article","type_order":1,"title":"Materiali didattici come oggetti digitali FAIR: una metodologia condivisa per la formazione in H2IOSC","year":2025,"authors":["Pedonese, G.","Frontini, F.","Ottaviani, R.","Boschetti, F.","Spadi, A.","Francalanci, L.","Scognamiglio, A.","Restaneo, P.","Chaban, A.","Striova, J.","Benassi, L."],"authors_source":"Pedonese, Giulia; Frontini, Francesca; Ottaviani, Roberta; Boschetti, Federico; Spadi, Alessia; Francalanci, Lucia; Scognamiglio, Alessia; Restaneo, Pietro; Chaban, Antonina; Striova, Jana; Benassi, Laura","authors_cnr_name":["PEDONESE, GIULIA","FRONTINI, FRANCESCA","OTTAVIANI, ROBERTA","BOSCHETTI, FEDERICO","SPADI, ALESSIA","FRANCALANCI, LUCIA","SCOGNAMIGLIO, ALESSIA","RESTANEO, PIETRO","CHABAN, ANTONINA","STRIOVA, JANA","BENASSI, LAURA"],"authors_cnr_id":["rp17300","rp02790","rp17076","rp04876","rp13683","rp17319","rp16531","rp11040","rp13702","rp16085","rp07896"],"authors_cnr_institute":[],"abstract":"Il presente lavoro dettaglia la strategia per lo sviluppo di iniziative di formazione nell\u2019ambito del progetto PNRR sviluppato dal CNR Humanities and cultural Heritage ItalianOpen Science Cloud(H2IOSC) e mira ad aprire alla comunit\u00e0 italiana di riferimento il processo di applicazione delle linee guida di design e di fruizione di moduli didattici che integrino l\u2019uso delle Infrastrutture di Ricerca. In particolare, il contributo si sofferma suglistandard condivisi per la descrizione dei materiali didattici come oggetti digitali FAIR al fine di massimizzarne il riutilizzo in un\u2019ottica train the trainers e sulla descrizione dei requisiti per l\u2019implementazione dell\u2019infrastruttura di training. Dopo aver descritto la strategia didattica (Sezione 2) e l\u2019applicazione della metodologia FAIR-by-Design di Skills4EOSC ai materiali didattici preesistenti (Sezione 3), il lavoro descrive il processo di ideazione di due piattaforme con funzionalit\u00e0 coerenti ai requisiti degli oggetti didattici prendendo ad esempio il corso CLARIN Introduction to Language Data: Standards and Repositoriestradotto e adattato in H2IOSC (Sezione 4)","keywords":["formazione, gestione dei dati, infrastrutture di ricerca, H2IOSC, principi FAIR"],"pages":"361-380","url":"https:\/\/iris.cnr.it\/handle\/20.500.14243\/552963","volume":"2025 (20)","doi":"10.6092\/issn.2532-8816\/21190","editors":[],"editors_source":"","published":"UMANISTICA DIGITALE","publisher":"","issn":"2532-8816","isbn":"","conference_name":"","conference_place":"","conference_date":"","last_updated_cnr":"2026-01-29 01:56:08","last_updated_oai":"2026-01-29 01:56:08","last_updated_www":"0000-00-00 00:00:00"},{"id":1228,"id_source":549328,"institutes":["ILC","ISPC"],"type":"book_chapter","type_order":4,"title":"Scienza aperta, dati e infrastrutture","year":2025,"authors":["Buscemi, F.","Frontini, F."],"authors_source":"Buscemi, F.; Frontini, F.","authors_cnr_name":["BUSCEMI, FRANCESCA","FRONTINI, FRANCESCA"],"authors_cnr_id":["rp01003","rp02790"],"authors_cnr_institute":[],"abstract":"Scienza Aperta, Dati e Infrastrutture. Impatto, impegno e prospettive degli Istituti del DSU rispetto a questi grandi temi, nel contesto nazionale ed europeo","keywords":["Scienza aperta","infrastrutture di ricerca","open publishing"],"pages":"34-44","url":"https:\/\/iris.cnr.it\/handle\/20.500.14243\/549328","volume":"","doi":"","editors":["Filippetti, A.","Sfameni, C.","Antonini, G."],"editors_source":"Filippetti A.; Sfameni C.; Antonini G.","published":"LE SCIENZE UMANE E SOCIALI NEL XXI SECOLO: COMPRENDERE E TRASFORMARE LA SOCIETA\u0300","publisher":"CNR Edizioni (Roma, ITA)","issn":"","isbn":"9788880807322","conference_name":"","conference_place":"Roma","conference_date":"","last_updated_cnr":"2025-12-10 01:53:13","last_updated_oai":"2025-12-10 01:53:13","last_updated_www":"0000-00-00 00:00:00"},{"id":1278,"id_source":562981,"institutes":["ILC","ISTI"],"type":"conference_article","type_order":7,"title":"Novel benchmark for NER in the wastewater and stormwater domain","year":2025,"authors":["Cardillo, F. A.","Debole, F.","Frontini, F.","Aelami, M.","Chahinian, N.","Conrad, S."],"authors_source":"Cardillo, F. A.; Debole, F.; Frontini, F.; Aelami, M.; Chahinian, N.; Conrad, S.","authors_cnr_name":["CARDILLO, FRANCO ALBERTO","DEBOLE, FRANCA","FRONTINI, FRANCESCA"],"authors_cnr_id":["rp02590","rp03513","rp02790"],"authors_cnr_institute":[],"abstract":"Efficient wastewater and stormwater management is mandatory for sustainable cities. Extracting structured knowledge from reports and regulations is challenging due to domain-specific terminology and multilingual contexts. This work focuses on domain-specific Named Entity Recognition (NER) as a first step towards effective relation and information extraction to support decision making. A multilingual benchmark is crucial for evaluating these methods. This study develops a French-Italian domain-specific text corpus for wastewater management. It evaluates state-of-the-art NER methods, including LLM-based approaches, to provide a reliable baseline for future strategies and explores automated annotation projection in view of an extension of the corpus to new languages","keywords":["Annotation projection","Domain-specific corpus","LLMs for NER","Multilingual NLP","Named Entity Recognition"],"pages":"226-231","url":"https:\/\/ieeexplore.ieee.org\/document\/11224095","volume":"","doi":"10.1109\/cist65886.2025.11224095","editors":[],"editors_source":"","published":"Cist 2025 proceedings","publisher":"Institute of Electrical and Electronics Engineers (USA)","issn":"","isbn":"979-8-3315-4384-6","conference_name":"Cist 2025-8th IEEE International Congress on Information Science and Technology","conference_place":"USA","conference_date":"","last_updated_cnr":"2026-01-16 01:15:20","last_updated_oai":"2026-01-16 01:15:20","last_updated_www":"0000-00-00 00:00:00"},{"id":1945,"id_source":571121,"institutes":["ILC","ILIESI","INO","ISPF","ISTI","OVI"],"type":"conference_article","type_order":7,"title":"Bridging Disciplines for Heritage Professionals: The H2IOSC Digital Training Platform (by CLARIN, DARIAH, E-RIHS and OPERAS)","year":2025,"authors":["Chaban, A.","Benassi, L.","Pedonese, G.","Frontini, F.","Ottaviani, R.","Boschetti, F.","Spadi, A.","Francalanci, L.","Sconamiglio, A.","Restaneo, P.","Striova, J."],"authors_source":"Chaban, Antonina; Benassi, Laura; Pedonese, Giulia; Frontini, Francesca; Ottaviani, Roberta; Boschetti, Federico; Spadi, Alessia; Francalanci, Lucia; Sconamiglio, Alessia; Restaneo, Pietro; Striova, Jana","authors_cnr_name":["CHABAN, ANTONINA","BENASSI, LAURA","PEDONESE, GIULIA","FRONTINI, FRANCESCA","OTTAVIANI, ROBERTA","BOSCHETTI, FEDERICO","SPADI, ALESSIA","FRANCALANCI, LUCIA","RESTANEO, PIETRO","STRIOVA, JANA"],"authors_cnr_id":["rp13702","rp07896","rp17300","rp02790","rp17076","rp04876","rp13683","rp17319","rp11040","rp16085"],"authors_cnr_institute":[],"abstract":"The complexity of the contemporary heritage field requires professionals to develop interdisciplinary skills and to collaborate across diverse disciplines, from social sciences and digital humanities to preservation of cultural heritage, archaeology and beyond. As digital and interactive tools become increasingly integrated into heritage studies, the training in the field is undergoing a significant transformation. The H2IOSC (Heritage and Humanities Italian Open Science Cloud) project has developed an innovative digital training infrastructure aimed at providing access to FAIR (Findable, Accessible, Interoperable and Reusable) courses and training materials. In this abstract we focus on the H2IOSC training platform, aimed to address the evolving training needs of our disciplinary communities. Created by H2IOSC WP8 (Work Package 8: Training, Engagement and Capacity Building) in collaboration with the E. T. T. S. p. A., it is maintained and hosted by CNR-ILC (Institute of Computational Linguistics \u201cA. Zampolli\u201d), host institution of CLARIN-IT, with the participation of the national nodes of DARIAH, E-RIHS, and OPERAS. The platform provides a flexible, customizable learning environment designed to support interdisciplinary education and continuous professional development in the Social Sciences, Digital Humanities and Cultural Heritage sectors. The platform features a comprehensive course catalogue containing training courses and modules developed or adapted by the four infrastructures within the H2IOSC project as FAIR training materials, designed for various target knowledge levels, from beginners to advanced learners in heritage and humanities fields. An intuitive dashboard allows users to access materials, track progress and use collaborative tools, including forums, chats, and virtual working groups for networking and communication. The platform incorporates quizzes, simulations, hands-on exercises, and gamification tools, making learning more interactive. Designed for accessibility and inclusivity, it is fully compatible with both desktop and mobile devices, enabling individual learning without geographical or temporal constraints. This paper outlines the development process of the platform, addressing the challenges encountered, the key achievements, and its potential to transform training within the Social Sciences, Digital Humanities and Heritage Sector. We highlight how the design of specialized digital heritage training materials has influenced the development of the Digital Asset Management (DAM) system, particularly its ability to support effective cataloguing, metadata management, and archival of multimedia data. Ultimately, we emphasize the importance of adopting best practices for interdisciplinary learning and explore how digital tools can foster greater collaboration and knowledge exchange across heritage and humanities disciplines","keywords":["heritage, digital humanities, training, FAIR"],"pages":"","url":"https:\/\/iris.cnr.it\/handle\/20.500.14243\/571121","volume":"","doi":"10.2312\/dh.20253036","editors":[],"editors_source":"","published":"Digital Heritage 2025","publisher":"","issn":"","isbn":"","conference_name":"","conference_place":"","conference_date":"","last_updated_cnr":"2026-03-05 01:38:52","last_updated_oai":"2026-03-05 01:38:52","last_updated_www":"0000-00-00 00:00:00"},{"id":1883,"id_source":570784,"institutes":["ILC"],"type":"conference_article","type_order":7,"title":"A Pilot Project for Promoting Linguistic Linked Open Data","year":2025,"authors":["Khan, A. F.","Mallia, M.","Quochi, V.","Pedonese, G.","Frontini, F.","Squadrito, E."],"authors_source":"Khan, Anas Fahad; Mallia, Michele; Quochi, Valeria; Pedonese, Giulia; Frontini, Francesca; Squadrito, Elisa","authors_cnr_name":["KHAN, ANAS FAHAD ASLAM","MALLIA, MICHELE","QUOCHI, VALERIA","PEDONESE, GIULIA","FRONTINI, FRANCESCA"],"authors_cnr_id":["rp05508","rp13895","rp13283","rp17300","rp02790"],"authors_cnr_institute":[],"abstract":"This paper presents a pilot initiative, part of the H2IOSC infrastructure, that strives to support and promote the creation, publication, and sharing of Linguistic Linked Open Data (LLOD) in Italy and beyond. We describe the different parts of the pilot project: those related to vocabulary hosting, RDF data publication, training development, and use case promotion. Key contributions include the publication and hosting of the REALITER series of lexicons, the PLLOD triple store platform, and LLOD-focused training initiatives. We also describe a series of use-cases taking place within the pilot","keywords":["Training,Linguistic Linked Open data, H2IOSC, CLARIN"],"pages":"1-6","url":"https:\/\/iris.cnr.it\/handle\/20.500.14243\/570784","volume":"","doi":"10.1109\/ieee-ch65308.2025.11279386","editors":[],"editors_source":"","published":"2025 IEEE International Conference on Cyber Humanities (IEEE-CH)","publisher":"","issn":"","isbn":"","conference_name":"2025 IEEE International Conference on Cyber Humanities (IEEE-CH)","conference_place":"","conference_date":"","last_updated_cnr":"2026-05-28 12:11:16","last_updated_oai":"2026-05-28 12:11:16","last_updated_www":"0000-00-00 00:00:00"},{"id":2080,"id_source":564161,"institutes":["ILC"],"type":"conference_article","type_order":7,"title":"Adapting UPSKILLS Learning Modules to the University Curricula. Best Practices and Lessons Learnt from the H2IOSC Training Experience at the University of Ferrara","year":2025,"authors":["Pedonese, G.","Frontini, F.","Del Fante, D.","Federici, E."],"authors_source":"Pedonese, Giulia; Frontini, Francesca; Del Fante, Dario; Federici, Eleonora","authors_cnr_name":["PEDONESE, GIULIA","FRONTINI, FRANCESCA"],"authors_cnr_id":["rp17300","rp02790"],"authors_cnr_institute":[],"abstract":"This paper details the steps taken to adapt and integrate the training materials developed by CLARIN ERIC in two bachelor\u2019s degree courses and one master\u2019s degree course at the University of Ferrara. The workflow applies the shared methodology developed within the Humanities and Heritage Italian Open Science Cloud project. It modifies the training materials of the UPSKILLS course \u201cIntroduction to Language Data: Standards and Repositories\u201d according to the needs of three target courses focusing on English to Italian translation: English Language Course for Tourism, English Language for Translation and English Language and Linguistics for Humanities, Arts and Archaeology. The result of this pilot is a documented example of how CLARIN services can be integrated into university teaching, including initial teacher training, and providing an opportunity to discuss the topic and a use case for trainers who intend to include CLARIN in their courses","keywords":["Training","Learning Resources","Language Data","FAIR principles","Research Infrastructures"],"pages":"37-47","url":"https:\/\/ecp.ep.liu.se\/index.php\/clarin\/article\/view\/1236","volume":"","doi":"10.3384\/ecp216.04","editors":["Vandeghinste, V.","Kontino, T."],"editors_source":"Vincent Vandeghinste and Thalassia Kontino","published":"Selected papers from the CLARIN Annual Conference 2024","publisher":"Linko\u0308ping University Electronic press' conference series (SWE)","issn":"","isbn":"978-91-8118-188-3","conference_name":"","conference_place":"SWE","conference_date":"","last_updated_cnr":"2026-03-04 01:27:47","last_updated_oai":"2026-03-04 01:27:47","last_updated_www":"0000-00-00 00:00:00"},{"id":1638,"id_source":552965,"institutes":["ILC","OVI","ISPF","ILIESI","INO"],"type":"conference_article","type_order":7,"title":"Dai Materiali Didattici alle Piattaforme FAIR: Costruire un\u2019Infrastruttura di Training in H2IOSC","year":2025,"authors":["Pedonese, G.","Frontini, F.","Ottaviani, R.","Boschetti, F.","Spadi, A.","Francalanci, L.","Scognamiglio, A.","Restaneo, P.","Chaban, A.","Striova, J.","Benassi, L."],"authors_source":"Pedonese, Giulia; Frontini, Francesca; Ottaviani, Roberta; Boschetti, Federico; Spadi, Alessia; Francalanci, Lucia; Scognamiglio, Alessia; Restaneo, Pietro; Chaban, Antonina; Striova, Jana; Benassi, Laura","authors_cnr_name":["PEDONESE, GIULIA","FRONTINI, FRANCESCA","OTTAVIANI, ROBERTA","BOSCHETTI, FEDERICO","SPADI, ALESSIA","FRANCALANCI, LUCIA","SCOGNAMIGLIO, ALESSIA","RESTANEO, PIETRO","CHABAN, ANTONINA","STRIOVA, JANA","BENASSI, LAURA"],"authors_cnr_id":["rp17300","rp02790","rp17076","rp04876","rp13683","rp17319","rp16531","rp11040","rp13702","rp16085","rp07896"],"authors_cnr_institute":[],"abstract":"Questo contributo si propone di illustrare la progettazione e lo sviluppo di un\u2019infrastruttura di formazione innovativa per le Scienze Umane e Sociali, basata sui principi FAIR e sulla promozione della Scienza Aperta, nell\u2019ambito del progetto Humanities and cultural Heritage Italian Open Science Cloud (H2IOSC). L\u2019obiettivo principale \u00e8 la creazione di un ecosistema integrato che renda i materiali didattici facilmente reperibili, accessibili, interoperabili e riutilizzabili. A tal fine, sono state implementate due piattaforme: H2IOSC Virtual Environment, dedicata all\u2019erogazione di corsi e risorse per studenti, e H2IOSC Training Library, un deposito per la conservazione e la condivisione di materiali didattici modulari. Entrambe le piattaforme si basano sulla metodologia \"FAIR-by-Design\" raccomandata dal progetto Skills4EOSC, che struttura il processo educativo in sei fasi, garantendo standard elevati di metadatazione e l\u2019uso di formati aperti. Con l\u2019implementazione di queste piattaforme, i cui servizi saranno resi disponibili a un livello di aggregazione pi\u00f9 alto nel Marketplace di H2IOSC, il progetto intende favorire un approccio scalabile e sostenibile alla formazione, promuovendo al contempo la collaborazione tra docenti e studenti","keywords":["formazione","gestione dei dati","infrastrutture di ricerca","principi FAIR","Scienza Aperta."],"pages":"473-477","url":"https:\/\/iris.cnr.it\/handle\/20.500.14243\/552965","volume":"","doi":"10.6092\/unibo\/amsacta\/8380","editors":[],"editors_source":"","published":"Diversita\u0300, Equita\u0300 e Inclusione: Sfide e Opportunita\u0300 per l\u2019Informatica Umanistica nell\u2019Era dell\u2019Intelligenza Artificiale, Proceedings del XIV Convegno Annuale AIUCD2025","publisher":"","issn":"","isbn":"978-88-942535-9-7","conference_name":"XIV Convegno Annuale AIUCD 2025","conference_place":"","conference_date":"","last_updated_cnr":"2026-01-29 01:54:59","last_updated_oai":"2026-01-29 01:54:59","last_updated_www":"0000-00-00 00:00:00"},{"id":304,"id_source":571001,"institutes":["ILC"],"type":"conference_article","type_order":7,"title":"From Collection to Transcription: A Workflow for Managing Speech Data by the CLARIN Trainers\u2019 Network","year":2025,"authors":["Van Den Heuvel, H.","Draxler, C.","Pedonese, G.","Frontini, F.","Van Der Lek, I."],"authors_source":"Henk van den Heuvel, Christoph Draxler, Giulia Pedonese, Francesca Frontini, Iulianna van der Lek","authors_cnr_name":[],"authors_cnr_id":[],"authors_cnr_institute":[],"abstract":"This proposal shares the experience of members of the CLARIN Trainers\u2019 Network in reusing, adapting and localizing existing learning content related to speech and oral data management to meet the learning needs of the CLARIN-IT research community in the context of the Humanities and Cultural Heritage Italian Open Science Cloud (H2IOSC) project","keywords":["training","FAIR data","Speech data","transcription chain"],"pages":"437-442","url":"https:\/\/ieeexplore.ieee.org\/servlet\/opac?punumber=11278902","volume":"","doi":"10.1109\/IEEE","editors":[],"editors_source":"","published":"Proceedings of the 2025 IEEE International Conference on Cyber Humanities (IEEE-CH) 8-10 September, Florence, Italy","publisher":"IEEE","issn":"","isbn":"979-8-3315-1435-8","conference_name":"IEEE International Conference on Cyber Humanities (IEEE-CH)","conference_place":"","conference_date":"","last_updated_cnr":"0000-00-00 00:00:00","last_updated_oai":"2026-03-05 01:36:25","last_updated_www":"0000-00-00 00:00:00"},{"id":246,"id_source":573982,"institutes":["ILC","ISTI","INO","OVI","ISPF","ILIESI"],"type":"technical_report","type_order":9,"title":"H2IOSC-D8. 1 Training Strategy","year":2025,"authors":["Benassi, L.","Boschetti, F.","Canova, L.","Chaban, A.","Degl'Innocenti, E.","Di Meo, C.","Frontini, F.","Monachini, M.","Ottaviani, R.","Pedonese, G.","Restaneo, P.","Scognamiglio, A.","Spadi, A.","Striova, J."],"authors_source":"Benassi, L.; Boschetti, F.; Canova, L.; Chaban, A.; Degl'Innocenti, E.; Di Meo, C.; Frontini, F.; Monachini, M.; Ottaviani, R.; Pedonese, G.; Restaneo, P.; Scognamiglio, A.; Spadi, A.; Striova, J.","authors_cnr_name":["BENASSI, LAURA","BOSCHETTI, FEDERICO","Canova, Leonardo","CHABAN, ANTONINA","DEGL'INNOCENTI, EMILIANO","DI MEO, CARMEN","FRONTINI, FRANCESCA","MONACHINI, MONICA","OTTAVIANI, ROBERTA","PEDONESE, GIULIA","RESTANEO, PIETRO","SCOGNAMIGLIO, ALESSIA","SPADI, ALESSIA","STRIOVA, JANA"],"authors_cnr_id":["rp07896","rp04876","rp15578","rp13702","rp09685","rp14423","rp02790","rp19457","rp17076","rp17300","rp11040","rp16531","rp13683","rp16085"],"authors_cnr_institute":[],"abstract":"In this document we describe the state of the art of the four participating infrastructures in terms of training and define the training strategy of the project H2IOSC in terms of overall vision, and along the following 3 lines:-Building the H2IOSC the training infrastructure with a training portal and a depositing service for training materials (technical and functional requirements are described for both)-Creating an offer of common H2IOSC training materials, aimed at facilitating the use of the H2IOSC marketplace and pilots-Strengthening the disciplinary offer for the four participating infrastructures This is a living document, evolving during the life of the project with the natural upgrade of the awareness of the needs expressed by the various groups interested (users already in the community and identified potential users)","keywords":["Training","Capacity building"],"pages":"49","url":"https:\/\/zenodo.org\/records\/14680109","volume":"","doi":"10.5281\/zenodo.14680108","editors":[],"editors_source":"","published":"","publisher":"","issn":"","isbn":"","conference_name":"","conference_place":"","conference_date":"","last_updated_cnr":"2026-04-12 07:04:04","last_updated_oai":"2026-04-12 07:04:04","last_updated_www":"0000-00-00 00:00:00"},{"id":1062,"id_source":561742,"institutes":["ILC"],"type":"misc","type_order":11,"title":"Introduzione alla Gestione dei Dati Orali","year":2025,"authors":["Van Den Heuvel, H.","Draxler, C.","Frontini, F.","Pedonese, G.","Van Der Lek, I."],"authors_source":"Van Den Heuvel, Henk; Draxler, Christoph; Frontini, Francesca; Pedonese, Giulia; Van Der Lek, Iulianna","authors_cnr_name":["FRONTINI, FRANCESCA","PEDONESE, GIULIA"],"authors_cnr_id":["rp02790","rp17300"],"authors_cnr_institute":[],"abstract":"Il corso affronta le tematiche legate alla gestione dei dati linguistici orali. Dopo un'introduzione generale alle possibilit\u00e0 offerte dall'infrastruttura CLARIN ERIC in fase di scoperta, raccolta e deposito di dati orali, si approfondiranno le questioni etico-legali connesse alla raccolta, gestione e conservazione dei dati e il procedimento di trascrizione automatica, con ulteriori possibilit\u00e0 di annotazione attraverso strumenti ti trattamento automatico del linguaggio. Il corso \u00e8 stato sviluppato con la collaborazione dei docenti della CLARIN Traners' Network nell'ambito della partecipazione di CLARIN-IT al Progetto H2IOSC-Humanities and cultural Heritage Italian Open Science Cloud finanziato dall\u2019Unione Europea NextGenerationEU \u2013 PNRR M4C2 \u2013 Codice progetto IR0000029 \u2013 CUP B63C22000730005. Il materiale si compone di tre unit\u00e0: Unit\u00e0 1-I Dati Linguistici Orali in CLARIN Questa unit\u00e0 fornisce una panoramica delle risorse e dei servizi offerti dall'Infrastruttura di Ricerca CLARIN ERIC a supporto della scoperta, dell'annotazione e del deposito dei dati linguistici orali in accordo con i principi FAIR e le buone pratiche della Scienza Aperta. Unit\u00e0 2-Raccolta e Gestione dei Dati Orali L'unit\u00e0 propone un'introduzione alle problematiche legate alla gestione dei dati orali dal punto di vista etico e legale. Gli aspetti legati al GDPR e alla normativa italiana di riferimento sono approfonditi in un gioco di ruolo interattivo. Unit\u00e0 3-Laboratorio di Trascrizione Automatica In questa unit\u00e0 interattiva, saranno affrontate le questioni relative ad alcuni strumenti e i software utili per la trascrizione dei dati. Si ringraziano le ricercatrici e i ricercatori impegnate\/i nel progetto PRIN Corpus SIM (Senecta Ipsa Morbus)-Spontaneous speech in healthy ageing per aver attivamente partecipato alle sessioni di didattica del 16 e 17 settembre 2024 presso l'Universit\u00e0 di Firenze, da cui \u00e8 stato tratto il materiale del corso","keywords":["Dati orali","Archivi orali","Trascrizione automatica"],"pages":"","url":"https:\/\/iris.cnr.it\/handle\/20.500.14243\/561742","volume":"","doi":"10.5281\/zenodo.17183051","editors":[],"editors_source":"","published":"","publisher":"","issn":"","isbn":"","conference_name":"","conference_place":"","conference_date":"","last_updated_cnr":"2026-03-04 01:27:42","last_updated_oai":"2026-03-04 01:27:42","last_updated_www":"0000-00-00 00:00:00"},{"id":1726,"id_source":475881,"institutes":["ILC"],"type":"journal_article","type_order":1,"title":"Toward a Representation of Semantic Change in Linked Data","year":2024,"authors":["Khan, A. F.","Frontini, F."],"authors_source":"Khan, Anas Fahad; Frontini, Francesca","authors_cnr_name":["KHAN, ANAS FAHAD ASLAM","FRONTINI, FRANCESCA"],"authors_cnr_id":["rp05508","rp02790"],"authors_cnr_institute":[],"abstract":"In this article, we introduce a new framework, the Intensional\u2013Ontological Model (IOM), for representing meaning, and especially for representing semantic change, in linguistic linked data resources. This framework, which makes use of previous work in the literature on lexical semantics and ontologies, is intended to help clarify what we mean when we model semantic change and to assist in elaborating different ontology patterns for doing so. In this work, we assume a simple architecture, one which is at the basis of the well-known OntoLex-Lemon vocabulary and which consists of one or more lexicons linked to an ontology. Our model, which is based on this architecture and informed by previous work on word senses and ontologies, is intended to provide a clear interpretation for the modelling of both onomasiological and semiasological changes, in both static and dynamic versions. This article describes how the IOM framework represents word meaning as the relationship between a word and an ontological concepts in the \u2019static\u2019 case, demonstrating that the IOM is compatible with OntoLex-Lemon (while at the same time providing a greater level of detail as to the meaning of the \u2019sense\u2019 and \u2019reference\u2019 relationships). It then goes on to detail how the IOM can help us understand how to model semantic shifts in linked data lexical resources with a focus on conceptual change and the addition of temporal information to semantic shift data","keywords":["linked data","semantic shift","ontologies","lexical semantics"],"pages":"","url":"https:\/\/iris.cnr.it\/handle\/20.500.14243\/475881","volume":"9 (6)","doi":"10.3390\/languages9060215","editors":[],"editors_source":"","published":"LANGUAGES","publisher":"","issn":"2226-471X","isbn":"","conference_name":"","conference_place":"","conference_date":"","last_updated_cnr":"2025-03-09 11:57:42","last_updated_oai":"2025-03-09 11:57:42","last_updated_www":"0000-00-00 00:00:00"},{"id":168,"id_source":475984,"institutes":["ILC"],"type":"book_chapter","type_order":4,"title":"Cartographie d\u2019une aventure Approche num\u00e9rique du Journal d\u2019un voyage fait aux Indes orientales de Robert Challe","year":2024,"authors":["Frontini, F.","Roth Boll, A.","Seguin, M. S."],"authors_source":"Frontini, Francesca; Francesca and, Roth-Boll; Amaury and, Seguin; Maria, Susana","authors_cnr_name":["FRONTINI, FRANCESCA"],"authors_cnr_id":["rp02790"],"authors_cnr_institute":[],"abstract":"L\u2019article propose d\u2019\u00e9tudier le Journal d\u2019un voyage de Robert Challe gr\u00e2ce aux outils des humanit\u00e9s num\u00e9riques. Nous avons reconstitu\u00e9 la cartographie de l\u2019aventure challienne et compar\u00e9 les trajets ainsi restitu\u00e9s avec leur exploitation viatique. La rencontre des espaces r\u00e9ellement visit\u00e9s avec leur repr\u00e9sentation textuelle fait ainsi \u00e9merger l\u2019existence d\u2019une forme de g\u00e9ographie m\u00e9morielle, affective et po\u00e9tique, indispensable au travail litt\u00e9raire de Robert Challe","keywords":["Contextualisation, humanite\u0301s nume\u0301riques, identification, localisation, marquage, me\u0301thodologie, observation, re\u0301fe\u0301rencement, spatialisation, technologie"],"pages":"","url":"https:\/\/iris.cnr.it\/handle\/20.500.14243\/475984","volume":"","doi":"10.48611\/isbn.978-2-406-16757-0.p.0247","editors":[],"editors_source":"","published":"Robert Challe et l\u2019aventure","publisher":"Classiques Garnier","issn":"","isbn":"978-2-406-16757-0","conference_name":"","conference_place":"","conference_date":"","last_updated_cnr":"2024-12-07 04:01:39","last_updated_oai":"2024-12-07 04:01:39","last_updated_www":"0000-00-00 00:00:00"},{"id":1104,"id_source":475921,"institutes":["ILC"],"type":"conference_article","type_order":7,"title":"MultiLexBATS: Multilingual Dataset of Lexical Semantic Relations","year":2024,"authors":["Gromann, D.","Goncalo Oliveira, H.","Pitarch, L.","Apostol, E. S.","Bernad, J.","Byty\u00e7i, E.","Cantone, C.","Carvalho, S.","Frontini, F.","Garabik, R.","Gracia, J.","Granata, L.","Khan, F.","Knez, T.","Labropoulou, P.","Liebeskind, C.","Pia Di Buono, M.","Ostro\u0161ki Ani\u0107, A.","Rackevi\u010dien\u0117, S.","Rodrigues, R.","S\u00e9rasset, G.","Selmistraitis, L.","Sidib\u00e9, M.","Silvano, P.","Spahiu, B.","Sogutlu, E.","Stankovi\u0107, R.","Truic\u0103, C. O.","Valunaite Oleskeviciene, G.","Zitnik, S.","Zdravkova, K."],"authors_source":"Gromann, Dagmar; Goncalo Oliveira, Hugo; Pitarch, Lucia; Apostol, Elena-Simona; Bernad, Jordi; Byty\u00e7i, Eliot; Cantone, Chiara; Carvalho, Sara; Frontini, Francesca; Garabik, Radovan; Gracia, Jorge; Granata, Letizia; Khan, Fahad; Knez, Timotej; Labropoulou, Penny; Liebeskind, Chaya; Pia Di Buono, Maria; Ostro\u0161ki Ani\u0107, Ana; Rackevi\u010dien\u0117, Sigita; Rodrigues, Ricardo; S\u00e9rasset, Gilles; Selmistraitis, Linas; Sidib\u00e9, Mahammadou; Silvano, Purifica\u00e7\u00e3o; Spahiu, Blerina; Sogutlu, Enriketa; Stankovi\u0107, Ranka; Truic\u0103, Ciprian-Octavian; Valunaite Oleskeviciene, Giedre; Zitnik, Slavko; Zdravkova, Katerina","authors_cnr_name":["FRONTINI, FRANCESCA","KHAN, ANAS FAHAD ASLAM"],"authors_cnr_id":["rp02790","rp05508"],"authors_cnr_institute":[],"abstract":"Understanding the relation between the meanings of words is an important part of comprehending natural language. Prior work has either focused on analysing lexical semantic relations in word embeddings or probing pretrained language models (PLMs), with some exceptions. Given the rarity of highly multilingual benchmarks, it is unclear to what extent PLMs capture relational knowledge and are able to transfer it across languages. To start addressing this question, we propose MultiLexBATS, a multilingual parallel dataset of lexical semantic relations adapted from BATS in 15 languages including low-resource languages, such as Bambara, Lithuanian, and Albanian. As experiment on cross-lingual transfer of relational knowledge, we test the PLMs{'} ability to (1) capture analogies across languages, and (2) predict translation targets. We find considerable differences across relation types and languages with a clear preference for hypernymy and antonymy as well as romance languages","keywords":["Lexical Semantic Relations","Multilingual Benchmark","BATS"],"pages":"11783-11793","url":"https:\/\/aclanthology.org\/2024.lrec-main.1029","volume":"","doi":"","editors":[],"editors_source":"","published":"Proceedings of the 2024 Joint International Conference on Computational Linguistics, Language Resources and Evaluation (LREC-COLING 2024)","publisher":"ELRA and ICCL","issn":"","isbn":"","conference_name":"","conference_place":"","conference_date":"","last_updated_cnr":"2024-12-07 04:02:34","last_updated_oai":"2024-12-07 04:02:34","last_updated_www":"0000-00-00 00:00:00"},{"id":745,"id_source":475941,"institutes":["ILC"],"type":"conference_article","type_order":7,"title":"CHAMU\u00c7A: Towards a Linked Data Language Resource of Portuguese Borrowings in Asian Languages","year":2024,"authors":["Khan, F.","Salgado, A.","Anuradha, I.","Costa, R.","Liyanage, C.","McCrae, J. P.","Ojha, A. K.","Rani, P.","Frontini, F."],"authors_source":"Khan, Fahad; Salgado, Ana; Anuradha, Isuri; Costa, Rute; Liyanage, Chamila; Mccrae, John P.; Ojha, Atul Kr.; Rani, Priya; Frontini, Francesca","authors_cnr_name":["KHAN, ANAS FAHAD ASLAM","FRONTINI, FRANCESCA"],"authors_cnr_id":["rp05508","rp02790"],"authors_cnr_institute":[],"abstract":"This paper presents the development of CHAMU\u00c7A, a novel lexical resource designed to document the influence of the Portuguese language on various Asian languages, with an initial focus on the languages of South Asia. Through the utilization of linked open data and the OntoLex vocabulary, CHAMU\u00c7A offers structured insights into the linguistic characteristics, and cultural ramifications of Portuguese borrowings across multiple languages. The article outlines CHAMU\u00c7A\u2019s potential contributions to the linguistic linked data community, emphasising its role in addressing the scarcity of resources for lesser-resourced languages and serving as a test case for organising etymological data in a queryable format. CHAMU\u00c7A emerges as an initiative towards the comprehensive catalogization and analysis of Portuguese borrowings, offering valuable insights into language contact dynamics, historical evolution, and cultural exchange in Asia, one that is based on linked data technology","keywords":["portuguese","ontolex","language contact","lexicon"],"pages":"","url":"https:\/\/aclanthology.org\/2024.ldl-1.6","volume":"","doi":"","editors":[],"editors_source":"","published":"Proceedings of the 9th Workshop on Linked Data in Linguistics @ LREC-COLING 2024","publisher":"ELRA and ICCL (Torino, Italia)","issn":"","isbn":"","conference_name":"","conference_place":"Torino, Italia","conference_date":"","last_updated_cnr":"2025-01-11 00:01:12","last_updated_oai":"2025-01-11 00:01:12","last_updated_www":"0000-00-00 00:00:00"},{"id":808,"id_source":506821,"institutes":["ILC","ILIESI","INO","ISPF","OVI"],"type":"conference_article","type_order":7,"title":"Materiali didattici come oggetti digitali FAIR: una metodologia condivisa per la formazione in H2IOSC","year":2024,"authors":["Pedonese, G.","Frontini, F.","Ottaviani, R.","Boschetti, F.","Spadi, A.","Francalanci, L.","Scognamiglio, A.","Restaneo, P.","Chaban, A.","Striova, J.","Benassi, L."],"authors_source":"Pedonese, Giulia; Frontini, Francesca; Ottaviani, Roberta; Boschetti, Federico; Spadi, Alessia; Francalanci, Lucia; Scognamiglio, Alessia; Restaneo, Pietro; Chaban, Antonina; Striova, Jana; Benassi, Laura","authors_cnr_name":["PEDONESE, GIULIA","FRONTINI, FRANCESCA","OTTAVIANI, ROBERTA","BOSCHETTI, FEDERICO","SPADI, ALESSIA","FRANCALANCI, LUCIA","SCOGNAMIGLIO, ALESSIA","RESTANEO, PIETRO","CHABAN, ANTONINA","STRIOVA, JANA","BENASSI, LAURA"],"authors_cnr_id":["rp17300","rp02790","rp17076","rp04876","rp13683","rp17319","rp16531","rp11040","rp13702","rp16085","rp07896"],"authors_cnr_institute":[],"abstract":"Il presente lavoro dettaglia la strategia per lo sviluppo di iniziative di formazione nell\u2019ambito del progetto H2IOSC e mira a coinvolgere la comunit\u00e0 italiana di riferimento sulle modalit\u00e0 di design e di fruizione di moduli didattici che integrino l\u2019uso delle Infrastrutture di Ricerca. In particolare, il contributo si sofferma sulla descrizione dei requisiti per l\u2019implementazione dell\u2019infrastruttura di training e sugli standard condivisi per la descrizione dei materiali didattici come oggetti digitali FAIR al fine di massimizzarne il riutilizzo in un\u2019ottica train the trainers","keywords":["Formazione","training","infrastrutture di ricerca","H2IOSC","principi FAIR."],"pages":"577-581","url":"https:\/\/amsacta.unibo.it\/id\/eprint\/7927\/","volume":"","doi":"10.6092\/unibo\/amsacta\/7927","editors":[],"editors_source":"","published":"Me. Te. Digitali. Mediterraneo in rete tra testi e contesti, Proceedings del XIII Convegno Annuale AIUCD2024","publisher":"","issn":"","isbn":"978-88-942535-8-0","conference_name":"XIII Convegno Annuale AIUCD2024","conference_place":"","conference_date":"","last_updated_cnr":"2026-01-29 01:56:44","last_updated_oai":"2026-01-29 01:56:44","last_updated_www":"0000-00-00 00:00:00"},{"id":1000,"id_source":475982,"institutes":["ILC"],"type":"technical_report","type_order":9,"title":"Data Stewardship Career Paths: Recommendations of the EOSC Task Force Data Stewardship Curricula and Career Paths","year":2024,"authors":["Kalov\u00e1, T.","Frontini, F.","Bracco, L.","Laetitia, D.","Meeus, J.","Hasani Mavriqi, I."],"authors_source":"Kalov\u00e1, Tereza; Frontini, Francesca; Bracco, Laetitia; Laetitia, Dunja; Meeus, Joke; Hasani-Mavriqi, Ilire","authors_cnr_name":["FRONTINI, FRANCESCA"],"authors_cnr_id":["rp02790"],"authors_cnr_institute":[],"abstract":"This document provides an overview of the topic of Data Stewardship Career Paths. Our review and summary of relevant reports and papers, ongoing initiatives, projects and surveys highlight the importance of ensuring more sustainable career paths for Data Stewards. The report further argues the need for further in-depth study and documentation of this topic. In particular, the analysis identifies relevant aspects that should be considered, ranging from employment conditions and salary to scientific recognition and roles. The report provides a list of recommendations and identifies activities that can be taken by the EOSC in the areas of Partnership, Association and Projects, summarised as follows: The EOSC Association and related projects should ensure that the overview of the current situation is kept up-to-date as a reference point. A section on the EOSC Association public website should be dedicated to Data Stewardship initiatives including a dedicated bibliography. To ensure long-term sustainability, a governing body should own and maintain this \"inventory\" as a point of reference; future projects should be encouraged to use it as a reference and provide input. The EOSC Association should ensure collaboration with international initiatives, in particular, the RDA IG Professionalizing Data Stewardship \u2013 TF Career Tracks and ensure coordination among current activities and studies carried out within the various EOSC Horizon Europe projects, and promote and support the organisation of dedicated events. The current and future EOSC projects and initiatives in the field should build on the work of this task force, as well as on the results of the studies above, and extend them by Applying qualitative methods such as guided interviews or focus groups to investigate the aspects covered in section 6 (\u201cData Stewardship Careers-What Counts\u201d) in more depthDeveloping Data Steward Personas based on the proposed methodology detailed in Annex 1The EOSC Partnership (EOSC Association, European Commission and Steering Board) should establish a permanent Data Stewardship expert group (including representatives from the various existing initiatives and this task force) with the following responsibilities: Develop and implement a monitoring framework that will allow the EOSC and other national and international institutions to support the career paths and development of personnel hired (at least in part) in Data Stewardship rolesAdvise the Association and the relevant (ongoing and future) projects and initiativesIssue recommendations on further activities of the Association regarding Data StewardshipIn consideration of the importance of the key role of Data Stewards in the development and implementation of the EOSC and to facilitate the exchange with and among Data Stewards, a further objective of the EOSC Association should be to assess the need for a professional network for Data Stewards, at least on the European level. &nbsp; The use of an innovative methodology, the Persona workshops, is recommended alongside further surveys and in-depth interviews to explore the relevant aspects of career paths and professional development of Data Stewards, including the roles and responsibilities of employers and Data Stewardship training programmes. &nbsp; Context This work was carried out within the Data Stewardship Curricula and Career Paths EOSC Task Force framework, particularly its Career Paths work stream","keywords":["career paths, data stewards, EOSC"],"pages":"","url":"https:\/\/zenodo.org\/records\/11077722","volume":"","doi":"10.5281\/zenodo.11077722","editors":[],"editors_source":"","published":"","publisher":"","issn":"","isbn":"","conference_name":"","conference_place":"","conference_date":"","last_updated_cnr":"2025-01-24 23:18:38","last_updated_oai":"2025-01-24 23:18:38","last_updated_www":"0000-00-00 00:00:00"},{"id":726,"id_source":483001,"institutes":["ILC","IGSG"],"type":"misc","type_order":11,"title":"Linguistically annotated multilingual comparable corpora of parliamentary debates ParlaMint. ana 4. 1","year":2024,"authors":["Erjavec, T.","Kopp, M.","Ogrodniczuk, M.","Osenova, P.","Agerri, R.","Agirrezabal, M.","Agnoloni, T.","Aires, J.","Albini, M.","Alkorta, J.","Antiba Cartazo, I.","Arrieta, E.","Barcala, M.","Bardanca, D.","Barkarson, S.","Bartolini, R.","Battistoni, R.","Bel, N.","Bonet Ramos, M. D. M.","Calzada P\u00e9rez, M.","Cardoso, A.","\u00c7\u00f6ltekin, \u00c7.","Coole, M.","Dar\u0123is, R.","De Does, J.","De Libano, R.","Depoorter, G.","Depuydt, K.","Diwersy, S.","Dod\u00e9, R.","Fernandez, K.","Fern\u00e1ndez Rei, E.","Frontini, F.","Garcia, M.","Garc\u00eda D\u00edaz, N.","Garc\u00eda Louzao, P.","Gavriilidou, M.","Gkoumas, D.","Grigorov, I.","Grigorova, V.","Haltrup Hansen, D.","Iruskieta, M.","Jarlbrink, J.","Jelencsik M\u00e1tyus, K.","Jongejan, B.","Kahusk, N.","Kirnbauer, M.","Kryvenko, A.","Ligeti Nagy, N.","Ljube\u0161i\u0107, N.","Luxardo, G.","Magari\u00f1os, C.","Magnusson, M.","Marchetti, C.","Marx, M.","Meden, K.","Mendes, A.","Mochtak, M.","M\u00f6lder, M.","Montemagni, S.","Navarretta, C.","Nito\u0144, B.","Nor\u00e9n, F. M.","Nwadukwe, A.","Ojster\u0161ek, M.","Pan\u010dur, A.","Papavassiliou, V.","Pereira, R.","P\u00e9rez Lago, M.","Piperidis, S.","Pirker, H.","Pisani, M.","Pol, H. V. D.","Prokopidis, P.","Quochi, V.","Rayson, P.","Regueira, X. L.","Rii, A.","Rudolf, M.","Ruisi, M.","Rupnik, P.","Schopper, D.","Simov, K.","Sinikallio, L.","Skubic, J.","Tamper, M.","Tungland, L. M.","Tuominen, J.","Van Heusden, R.","Varga, Z.","V\u00e1zquez Abu\u00edn, M.","Venturi, G.","Vidal Migu\u00e9ns, A.","Vider, K.","Vivel Couso, A.","Vladu, A. I.","Wissik, T.","Yrj\u00e4n\u00e4inen, V.","Zevallos, R.","Fi\u0161er, D."],"authors_source":"Erjavec, Toma\u017e; Kopp, Maty\u00e1\u0161; Ogrodniczuk, Maciej; Osenova, Petya; Agerri, Rodrigo; Agirrezabal, Manex; Agnoloni, Tommaso; Aires, Jos\u00e9; Albini, Monica; Alkorta, Jon; Antiba-Cartazo, Iv\u00e1n; Arrieta, Ekain; Barcala, Mario; Bardanca, Daniel; Barkarson, Starka\u00f0ur; Bartolini, Roberto; Battistoni, Roberto; Bel, Nuria; Bonet Ramos, Maria del Mar; Calzada P\u00e9rez, Mar\u00eda; Cardoso, Aida; \u00c7\u00f6ltekin, \u00c7a\u011fr\u0131; Coole, Matthew; Dar\u0123is, Roberts; de Does, Jesse; de Libano, Ruben; Depoorter, Griet; Depuydt, Katrien; Diwersy, Sascha; Dod\u00e9, R\u00e9ka; Fernandez, Kike; Fern\u00e1ndez Rei, Elisa; Frontini, Francesca; Garcia, Marcos; Garc\u00eda D\u00edaz, Noelia; Garc\u00eda Louzao, Pedro; Gavriilidou, Maria; Gkoumas, Dimitris; Grigorov, Ilko; Grigorova, Vladislava; Haltrup Hansen, Dorte; Iruskieta, Mikel; Jarlbrink, Johan; Jelencsik-M\u00e1tyus, Kinga; Jongejan, Bart; Kahusk, Neeme; Kirnbauer, Martin; Kryvenko, Anna; Ligeti-Nagy, No\u00e9mi; Ljube\u0161i\u0107, Nikola; Luxardo, Giancarlo; Magari\u00f1os, Carmen; Magnusson, M\u00e5ns; Marchetti, Carlo; Marx, Maarten; Meden, Katja; Mendes, Am\u00e1lia; Mochtak, Michal; M\u00f6lder, Martin; Montemagni, Simonetta; Navarretta, Costanza; Nito\u0144, Bart\u0142omiej; Nor\u00e9n, Fredrik Mohammadi; Nwadukwe, Amanda; Ojster\u0161ek, Mihael; Pan\u010dur, Andrej; Papavassiliou, Vassilis; Pereira, Rui; P\u00e9rez Lago, Mar\u00eda; Piperidis, Stelios; Pirker, Hannes; Pisani, Marilina; Pol, Henk van der; Prokopidis, Prokopis; Quochi, Valeria; Rayson, Paul; Regueira, Xos\u00e9 Lu\u00eds; Rii, Andriana; Rudolf, Micha\u0142; Ruisi, Manuela; Rupnik, Peter; Schopper, Daniel; Simov, Kiril; Sinikallio, Laura; Skubic, Jure; Tamper, Minna; Tungland, Lars Magne; Tuominen, Jouni; van Heusden, Ruben; Varga, Zs\u00f3fia; V\u00e1zquez Abu\u00edn, Marta; Venturi, Giulia; Vidal Migu\u00e9ns, Adri\u00e1n; Vider, Kadri; Vivel Couso, Ainhoa; Vladu, Adina Ioana; Wissik, Tanja; Yrj\u00e4n\u00e4inen, V\u00e4in\u00f6; Zevallos, Rodolfo; Fi\u0161er, Darja","authors_cnr_name":["AGNOLONI, TOMMASO","BARTOLINI, ROBERTO","FRONTINI, FRANCESCA","MONTEMAGNI, SIMONETTA","QUOCHI, VALERIA","VENTURI, GIULIA"],"authors_cnr_id":["rp21506","rp00239","rp02790","rp16780","rp13283","rp00732"],"authors_cnr_institute":[],"abstract":"ParlaMint 4. 1 is a set of comparable corpora containing transcriptions of parliamentary debates of 29 European countries and autonomous regions, mostly starting in 2015 and extending to mid-2022. The individual corpora comprise between 9 and 126 million words and the complete set contains over 1. 2 billion words. The transcriptions are divided by days with information on the term, session and meeting, and contain speeches marked by the speaker and their role (e. g. chair, regular speaker). The speeches also contain marked-up transcriber comments, such as gaps in the transcription, interruptions, applause, etc. The corpora have extensive metadata, most importantly on speakers (name, gender, MP and minister status, party affiliation), on their political parties and parliamentary groups (name, coalition\/opposition status, Wikipedia-sourced left-to-right political orientation, and CHES variables, https: \/\/www. chesdata. eu\/). Note that some corpora have further metadata, e. g. the year of birth of the speakers, links to their Wikipedia articles, their membership in various committees, etc. The transcriptions are also marked with the subcorpora they belong to (\"reference\", until 2020-01-30, \"covid\", from 2020-01-31, and \"war\", from 2022-02-24). An overview of the statistics of the corpora is avaialable on GitHub in the folder Build\/Metadata, in particular for the release 4. 1 at https: \/\/github. com\/clarin-eric\/ParlaMint\/tree\/v4. 1\/Build\/Metadata. The corpora are encoded according to the ParlaMint encoding guidelines (https: \/\/clarin-eric. github. io\/ParlaMint\/) and schemas (included in the distribution). The ParlaMint. ana linguistic annotation includes tokenization; sentence segmentation; lemmatisation; Universal Dependencies part-of-speech, morphological features, and syntactic dependencies; and the 4-class CoNLL-2003 named entities. Some corpora also have further linguistic annotations, in particular PoS tagging according a language-specific scheme, with their corpus TEI headers giving further details on the annotation vocabularies and tools used. This entry contains the ParlaMint. ana TEI-encoded linguistically annotated corpora; the derived CoNLL-U files along with TSV metadata of the speeches; and the derived vertical files (with their registry file), suitable for use with CQP-based concordancers, such as CWB, noSketch Engine or KonText. Also included is the 4. 1 release of the sample data and scripts available at the GitHub repository of the ParlaMint project at https: \/\/github. com\/clarin-eric\/ParlaMint and the log files produced in the process of building the corpora for this release. The log files show e. g. known errors in the corpora, while more information about known problems is available in the open issues at the GitHub repository of the project. This entry contains the linguistically marked-up version of the corpus, while the text version, i. e. without the linguistic annotation is also available at http: \/\/hdl. handle. net\/11356\/1912. Another related resource, namely the ParlaMint corpora machine translated to English ParlaMint-en. ana 4. 1 can be found at http: \/\/hdl. handle. net\/11356\/1910. As opposed to the previous version 4. 0, this version fixes a number of bugs and restructures the ParlaMint GitHub repository. The DK corpus has been linguistically re-annotated to remove bugs, while its speeches are now also marked with topics. The PT corpus has been extended to 2024-03 and the UA corpus to 2023-11, which also has improved language marking (uk vs. ru) on segments","keywords":["ParlaCLARIN, linguistic annotation, pos-tagging, Named Entity Recognition, linguistic dependency annotation, UD"],"pages":"","url":"https:\/\/iris.cnr.it\/handle\/20.500.14243\/483001","volume":"","doi":"","editors":[],"editors_source":"","published":"","publisher":"","issn":"","isbn":"","conference_name":"","conference_place":"","conference_date":"","last_updated_cnr":"2025-03-07 06:08:51","last_updated_oai":"2025-03-07 06:08:51","last_updated_www":"0000-00-00 00:00:00"},{"id":522,"id_source":561741,"institutes":["ILC"],"type":"misc","type_order":11,"title":"Linguistic Linked Open Data for Humanists","year":2024,"authors":["Pedonese, G.","Khan, A. F.","Mallia, M.","Frontini, F.","Quochi, V.","Squadrito, E."],"authors_source":"Pedonese, Giulia; Khan, Anas Fahad; Mallia, Michele; Frontini, Francesca; Quochi, Valeria; Squadrito, Elisa","authors_cnr_name":["PEDONESE, GIULIA","KHAN, ANAS FAHAD ASLAM","MALLIA, MICHELE","FRONTINI, FRANCESCA","QUOCHI, VALERIA"],"authors_cnr_id":["rp17300","rp05508","rp13895","rp02790","rp13283"],"authors_cnr_institute":[],"abstract":"Having achieved popularity as a way of publishing and accessing data in different fields of the sciences and for sharing large encyclopaedic datasets such as DBpedia (derived from Wikipedia), linked data is becoming more and more popular in different areas of the humanities. In this course we will present a comprehensive introduction to the creation, publication, and use of linked open data for anyone who wants to work with linguistic datasets \u2013 such as lexicons and corpora \u2013 and especially for those who come from a linguistic or humanist background. We will look at the basics of linked data and the Semantic Web and introduce the various different standards technologies that make up the Semantic Web stack before focusing on the particular case of linked data language resources. During the course we will study the most important tools, vocabularies, and resources available in the Semantic Web and provide hands-on training for the creation and querying of linguistic linked data. We will look at how Semantic Web technologies can contribute to the creation of FAIR language resources as well as how to publish your resource on the linked open data cloud. We will also show how the Semantic Web query language SPARQL can be a powerful tool for data exploration","keywords":["Linked Open Data","Linguistics"],"pages":"","url":"https:\/\/iris.cnr.it\/handle\/20.500.14243\/561741","volume":"","doi":"10.5281\/zenodo.13897931","editors":[],"editors_source":"","published":"","publisher":"","issn":"","isbn":"","conference_name":"","conference_place":"","conference_date":"","last_updated_cnr":"2026-03-04 01:28:03","last_updated_oai":"2026-03-04 01:28:03","last_updated_www":"0000-00-00 00:00:00"},{"id":1946,"id_source":561743,"institutes":["ILC"],"type":"misc","type_order":11,"title":"Introduzione ai Dati Linguistici: Standard e Archivi Digitali","year":2024,"authors":["Van Der Lek, I.","Fi\u0161er, D.","Frontini, F.","Pedonese, G."],"authors_source":"Van Der Lek, Iulianna; Fi\u0161er, Darja; Frontini, Francesca; Pedonese, Giulia","authors_cnr_name":["FRONTINI, FRANCESCA","PEDONESE, GIULIA"],"authors_cnr_id":["rp02790","rp17300"],"authors_cnr_institute":[],"abstract":"Il corso \"Introduzione ai Dati Linguistici: Standard e Archivi Digitali\" introduce gli insegnanti e gli studenti all'uso degli archivi digitali di dati della ricerca e al loro ruolo nel ciclo di vita dei dati linguistici nel contesto degli principi FAIR e delle buone pratiche della Scienza Aperta. I materiali del corso sono suddivisi in unit\u00e0 e sono intesi come contenuti didattici per i docenti che insegnano a livello di laurea triennale o laurea magistrale, che sono invitati a sfogliare i materiali, esportarli per l'uso nel Learning Management System della loro istituzione e adattarli ai propri scopi come ritengono opportuno. Questo corso traduce in italiano e aggiorna i materiali di: van der Lek, Iulianna; Fi\u0161er, Darja. (2023). Introduction to Language Data: Standards and Repositories. In UPSKILLS Learning Content. https: \/\/upskillsproject. eu\/project\/standards_repositories\/. CC BY 4. 0. https: \/\/creativecommons. org\/licenses\/by\/4. 0\/ L'adattamento si \u00e8 svolto nell'ambito del progetto Humanities and cultural Heritage Italian Open Science Cloud (https: \/\/www. h2iosc. cnr. it\/), Work Package 8 \"Training, Capacity Building, Engagement\", a cura del personale CNR-ILC dedicato all'Attivit\u00e0 8. 2 \"Teach CLARIN, Teach with CLARIN\". Progetto H2IOSC-Humanities and cultural Heritage Italian Open Science Cloud finanziato dall\u2019Unione Europea NextGenerationEU \u2013 PNRR M4C2 \u2013 Codice progetto IR0000029 \u2013 CUP B63C22000730005","keywords":["Dati Linguistici","Gestione dati"],"pages":"","url":"https:\/\/iris.cnr.it\/handle\/20.500.14243\/561743","volume":"","doi":"10.5281\/zenodo.13911935","editors":[],"editors_source":"","published":"","publisher":"","issn":"","isbn":"","conference_name":"","conference_place":"","conference_date":"","last_updated_cnr":"2026-03-04 01:28:17","last_updated_oai":"2026-03-04 01:28:17","last_updated_www":"0000-00-00 00:00:00"},{"id":912,"id_source":476001,"institutes":["ILC"],"type":"journal_article","type_order":1,"title":"Tool criticism in practice. On methods, tools and aims of computational literary studies","year":2023,"authors":["Berenike Herrmann, J.","Bories, A. S.","Frontini, F.","Jacquot, C.","Pielstr\u00f6m, S.","Rebora, S.","Rockwell, G.","Sinclair, S."],"authors_source":"Berenike Herrmann, J.; Bories, Anne-Sophie; Frontini, Francesca; Jacquot, Cl\u00e9mence; Pielstr\u00f6m, Steffen; Rebora, Simone; Rockwell, Geoffrey; Sinclair, St\u00e9fan","authors_cnr_name":["FRONTINI, FRANCESCA"],"authors_cnr_id":["rp02790"],"authors_cnr_institute":[],"abstract":"This paper is a case-driven contribution to the discussion on the method-theory relationship in practices within the field of Computational Literary Studies (CLS). Progress in this field dedicated to the computational analysis of literary texts has long revolved around the new, digital tools: tools, as computational devices for analysis, have had here a comparatively strong status as research entities of their own, while their ontological status has remained unclear to the day. As a rule, they have widely been imported from the fields of data science and NLP, while less often being hand-tailored to specific tasks within interdisciplinary settings. Although studies within CLS are evolving to both a higher degree of specialization in method (going beyond the limitations of out-of-the-box tools) and a stronger theoretical modeling, the technological dimension remains a defining factor. An unreflective adoption of technology in the shape of tools can compromise the plausibility and the reproducibility of the results produced using these tools. Our paper presents a multi-faceted intervention to the discussion around tools, methods, and the research questions that are answered with them. It presents research perspectives first conceived at the ADHO SIG-DLS workshop Anatomy of tools: A closer look at textual DH methodologies that took place in Utrecht in July 2019. At that event, the authors discussed selected case studies to address tool criticism from several angles. Our goal was to leverage a tool-critical perspective, in order to \u201ctake stock, reflect upon and critically comment upon our own practices\u201d within CLS. We identified Textom\u00e9trie, Stylometry, and Semantic Text Mining as three central types of hands-on CLS. For each of these sub-fields, we asked: What are our tools and methods-in-use? What are the implications of using a tool-oriented perspective as opposed to a methodology-oriented one? How do either relate to research questions and theory? These questions were explored by case-studies on an exemplary basis. The unifying perspective of this paper is an applied tool criticism \u2013 a critical inquiry leveraged towards crucial dimensions of CLS practices. Here we re-compose the original oral papers and add entirely new sections to it, to create a useful overview of the issue through a combination of perspectives. While we elaborated the thematic connections between the individual case studies, we hope the interactive spirit of an exemplary exchange remains palpable: individual research perspectives shape the case studies reported for Textom\u00e9trie, Stylometry and Semantic Text Mining, are complemented by further studies showcasing CLS-specific perspectives on replicability and domain-specific research, and a short section discussing a tool inventory as a practical, community-based incarnation of tool criticism. The article reflects thus a rich array of perspectives on tool criticism, including the complementary perspective of tool defense \u2013 arguing that we need tools and methods as a basic common ground on how to carry out fundamental operations of analysis and interpretation within a community","keywords":["tool criticism, digital literary studies, digital humanities"],"pages":"","url":"https:\/\/www.digitalhumanities.org\/dhq\/vol\/17\/2\/000687\/000687.html","volume":"017 (2)","doi":"","editors":[],"editors_source":"","published":"DIGITAL HUMANITIES QUARTERLY","publisher":"","issn":"1938-4122","isbn":"","conference_name":"","conference_place":"","conference_date":"","last_updated_cnr":"2025-03-01 07:49:17","last_updated_oai":"2025-03-01 07:49:17","last_updated_www":"0000-00-00 00:00:00"},{"id":378,"id_source":475961,"institutes":["ILC"],"type":"journal_article","type_order":1,"title":"The CLARIN infrastructure as an interoperable language technology platform for SSH and beyond","year":2023,"authors":["Branco, A.","Eskevich, M.","Frontini, F.","Hajic, J.","Hinrichs, E.","Jong, F.","Kamocki, P.","Konig, A.","Linden, K.","Navarretta, C.","Piasecki, M.","Piperidis, S.","Pitkanen, O.","Simov, K.","Skadina, I.","Trippel, T.","Witt, A.","Zinn, C."],"authors_source":"Branco, A.; Eskevich, M.; Frontini, F.; Hajic, J.; Hinrichs, E.; Jong, F.; Kamocki, P.; Konig, A.; Linden, K.; Navarretta, C.; Piasecki, M.; Piperidis, S.; Pitkanen, O.; Simov, K.; Skadina, I.; Trippel, T.; Witt, A.; Zinn, C.","authors_cnr_name":["FRONTINI, FRANCESCA"],"authors_cnr_id":["rp02790"],"authors_cnr_institute":[],"abstract":"CLARIN is a European Research Infrastructure Consortium developing and providing a federated and interoperable platform to support scientists in the field of the Social Sciences and Humanities in carrying-out language-related research. This contribution provides an overview of the entire infrastructure with a particular focus on tool interoperability, ease of access to research data, tools and services, the importance of sharing knowledge within and across (national) communities, and community building. By taking into account FAIR principles from the very beginning, CLARIN succeeded in becoming a successful example of a research infrastructure that is actively used by its members. The benefits CLARIN members reap from their infrastructure secure a future for their common good that is both sustainable and attractive to partners beyond the original target groups","keywords":["Interoperability","Language resources","Language technology","Research infrastructure","Social sciences and humanities"],"pages":"","url":"https:\/\/iris.cnr.it\/handle\/20.500.14243\/475961","volume":"","doi":"10.1007\/s10579-023-09658-z","editors":[],"editors_source":"","published":"LANGUAGE RESOURCES AND EVALUATION","publisher":"","issn":"1574-020X","isbn":"","conference_name":"","conference_place":"","conference_date":"","last_updated_cnr":"2025-06-12 01:36:33","last_updated_oai":"2025-06-12 01:36:33","last_updated_www":"0000-00-00 00:00:00"},{"id":1691,"id_source":469901,"institutes":["ILC"],"type":"book_chapter","type_order":4,"title":"Migrazione di testi e di codici manoscritti: risorse digitali per la ricostruzione dell'Occitania medievale","year":2023,"authors":["Caiti Russo, G.","Frontini, F."],"authors_source":"Caiti-Russo, Gilda; Frontini, Francesca","authors_cnr_name":["FRONTINI, FRANCESCA"],"authors_cnr_id":["rp02790"],"authors_cnr_institute":[],"abstract":"This paper offers an overview of the current panorama of digital language resources and approaches to the study of Medieval Occitan. Starting from a scattered tradition of witnesses, Occitan manuscripts are now being digitised in various projects, but not enough has been done to adopt interoperable and shared practices. By drawing from other philological traditions, and in particular the Digital Classics, we trace an inventory of best practices, tools, standards and formats that Digital Occitan Studies needs to develop in order to offer scholars access to Virtual Research environments of linked and interconnected language resources","keywords":["Occitan, Digital Philology, Digital Editions, Virtual Research Environments","Occitano, filologia digitale, edizioni digitali, ambienti virtuali di ricerca"],"pages":"537-568","url":"https:\/\/iris.cnr.it\/handle\/20.500.14243\/469901","volume":"","doi":"10.7410\/1678","editors":[],"editors_source":"","published":"Storie di idee nell'Europa mediterranea: trasmissione di parole e saperi nel Medioevo e nella prima eta\u0300 moderna","publisher":"Casalini-ISEM-Istituto di Storia dell'Europa Mediterranea (Cagliari)","issn":"","isbn":"978-88-97317-81-4","conference_name":"","conference_place":"Cagliari","conference_date":"","last_updated_cnr":"2025-02-08 06:22:42","last_updated_oai":"2025-02-08 06:22:42","last_updated_www":"0000-00-00 00:00:00"},{"id":1981,"id_source":476003,"institutes":["ILC"],"type":"book_chapter","type_order":4,"title":"The perception of voice qualities in audiobooks in the context of teaching French as a second language","year":2023,"authors":["Drengubiak, J.","Hirsch, F.","Didirkov\u00e1, I.","Frontini, F."],"authors_source":"Drengubiak, J\u00e1n; Hirsch, Fabrice; Didirkov\u00e1, Ivana; Frontini, Francesca","authors_cnr_name":["FRONTINI, FRANCESCA"],"authors_cnr_id":["rp02790"],"authors_cnr_institute":[],"abstract":"Audiobooks are a common part of everyday life and their possibilities are also used when learning a foreign language. The article comprehensively addresses the issue of audiobooks as a variation of the Listening comprehension activity on a sample of secondary and university students. In addition to their experience with audiobooks, their attitude to listening comprehension, it examines the preferences regarding specific characteristics of the voices of the narrators. The students first determined the perceived characteristic on a scale of 1-4 in the following categories: pitch (high-low), speed (slow-fast), melodicity (monotonous-melodious), articulation (comprehensible-incomprehensible). Later, subjective ratings were assigned to these categories on a scale from strong like to strong dislike. The preference research was conducted on an excerpt of Grand Meaulnes by A. Fournier, which was recorded by two professional and two amateur narrators","keywords":[],"pages":"","url":"https:\/\/www.pulib.sk\/web\/pdf\/web\/viewer.html?file=\/web\/kniznica\/elpub\/dokument\/Drengubiak4\/subor\/9788055531786.pdf","volume":"","doi":"","editors":[],"editors_source":"","published":"Literatu\u0301ra vo vy\u0301uc\u030cbe \u2013 vyuc\u030covat\u030c literatu\u0301ru; La litte\u0301rature dans l'enseignement-Enseigner la litte\u0301rature; Literature in teaching-Teaching literature","publisher":"","issn":"","isbn":"978-80-555-3178-6","conference_name":"","conference_place":"","conference_date":"","last_updated_cnr":"2025-01-24 02:14:45","last_updated_oai":"2025-01-24 02:14:45","last_updated_www":"0000-00-00 00:00:00"},{"id":1537,"id_source":475983,"institutes":["ILC"],"type":"book_chapter","type_order":4,"title":"Infrastrutture digitali per le scienze umane e sociali","year":2023,"authors":["Monachini, M.","Frontini, F."],"authors_source":"Monachini, Monica; Frontini, Francesca","authors_cnr_name":["MONACHINI, MONICA","FRONTINI, FRANCESCA"],"authors_cnr_id":["rp19457","rp02790"],"authors_cnr_institute":[],"abstract":"Questo capitolo esplora il ruolo delle infrastrutture di ricerca (IR) nel promuovere la collaborazione interdisciplinare e l\u2019innovazione tecnologica nell\u2019ambito delle Digital Humanities. Le IR, come CLARIN e DARIAH, rappresentano nodi cruciali per l\u2019accesso e la condivisione di risorse linguistiche, tecnologie e dati. Attraverso una panoramica dei servizi offerti, tra cui l\u2019archiviazione, l\u2019accesso federato e le applicazioni web, viene evidenziata la loro importanza per garantire principi FAIR (Findable, Accessible, Interoperable, Reusable) e supportare la ricerca umanistica e sociale in Europa. Il lavoro sottolinea l\u2019impatto strategico delle IR nel favorire la condivisione delle conoscenze e l\u2019integrazione delle risorse a livello transnazionale, contribuendo alla costruzione di ecosistemi di ricerca avanzati","keywords":["Infrastrutture di ricerca, Digital Humanities, Principi FAIR"],"pages":"197-213","url":"https:\/\/iris.cnr.it\/handle\/20.500.14243\/475983","volume":"DIGITAL HUMANITIES. METODI, STRUMENTI, SAPERI","doi":"","editors":[],"editors_source":"","published":"Digital {Humanities}. {Metodi}, strumenti, saperi","publisher":"Carocci Editore","issn":"","isbn":"978-88-290-1843-7","conference_name":"","conference_place":"","conference_date":"","last_updated_cnr":"2025-01-30 01:40:55","last_updated_oai":"2025-01-30 01:40:55","last_updated_www":"0000-00-00 00:00:00"},{"id":1518,"id_source":504563,"institutes":["ILC"],"type":"conference_article","type_order":7,"title":"CLARIN-IT: texts, documents and new contexts","year":2023,"authors":["Boschetti, F.","Del Grosso, A. M.","Del Gratta, R.","Frontini, F.","Monachini, M."],"authors_source":"Boschetti, Federico; DEL GROSSO, ANGELO MARIO; DEL GRATTA, Riccardo; Frontini, Francesca; Monachini, Monica","authors_cnr_name":["BOSCHETTI, FEDERICO","DEL GROSSO, ANGELO MARIO","DEL GRATTA, RICCARDO","FRONTINI, FRANCESCA","MONACHINI, MONICA"],"authors_cnr_id":["rp04876","rp02681","rp00284","rp02790","rp19457"],"authors_cnr_institute":[],"abstract":"In recent years, CLARIN has increasingly broadened its interest from linguistic resources to textual resources relevant to digital humanists. This new and attractive scenario requires new technologies for texts, variants, and digital representations of primary sources, their contexts, and complex relationships. VeDPH in Venice, CNR-ILC-CoPhiLab, and ILC4CLARIN in Pisa collaborate on DH projects. Together, they are working on extracting text from manuscript page images, annotating historical graffiti on georeferenced images, and identifying text in digital images of paintings and sculptures","keywords":["Research Infrastructure"],"pages":"53-56","url":"https:\/\/iris.cnr.it\/handle\/20.500.14243\/504563","volume":"","doi":"","editors":[],"editors_source":"","published":"CLARIN Annual Conference Proceedings 2023","publisher":"","issn":"","isbn":"","conference_name":"CLARIN Annual Conference Proceedings 2023","conference_place":"","conference_date":"","last_updated_cnr":"2024-12-18 04:46:39","last_updated_oai":"2024-12-18 04:46:39","last_updated_www":"0000-00-00 00:00:00"},{"id":449,"id_source":476002,"institutes":["ILC"],"type":"conference_article","type_order":7,"title":"ISO LMF 24613-6: A Revised Syntax Semantics Module for the Lexical Markup Framework","year":2023,"authors":["Frontini, F.","Romary, L.","Khan, A. F. A."],"authors_source":"Frontini, Francesca; Romary, Laurent; Khan, ANAS FAHAD ASLAM","authors_cnr_name":["FRONTINI, FRANCESCA","KHAN, ANAS FAHAD ASLAM"],"authors_cnr_id":["rp02790","rp05508"],"authors_cnr_institute":[],"abstract":"The Lexical Markup Framework (LMF) is a meta-model for representing data in monolingual and multilingual lexical databases with a view to its use in computer applications. The \"new LMF\" replaces the old LMF standard, ISO 24613: 2008, and is being published as a multi-part standard. This short paper introduces one of these new parts, ISO 24613-6, namely the Syntax and Semantics (SynSem) module. The SynSem module allows for the description of syntactic and semantic properties of lexemes, as well as the complex interactions between them. While the new standard remains faithful to (and backwards compatible with) the syntax and semantics coverage of the previous model, the new standard clarifies and simplifies it in a few places, which will be illustrated","keywords":["ISO, LMF, TEI, Semantics, Syntax"],"pages":"","url":"https:\/\/inria.hal.science\/hal-04117132","volume":"","doi":"","editors":[],"editors_source":"","published":"Proceedings of LDK 2023 \u2013 4th Conference on Language, Data and Knowledge","publisher":"","issn":"","isbn":"","conference_name":"","conference_place":"","conference_date":"","last_updated_cnr":"2025-01-24 23:18:11","last_updated_oai":"2025-01-24 23:18:11","last_updated_www":"0000-00-00 00:00:00"},{"id":1571,"id_source":475981,"institutes":["ILC"],"type":"conference_misc","type_order":8,"title":"Constructing an Old English WordNet: The Case of Guilt","year":2023,"authors":["Khan, A. F. A.","Cavallaro, M.","Cruz Gonz\u00e1lez, R.","D\u00edaz Vera, J.","Frontini, F.","Javier Minaya G\u00f3mez, F."],"authors_source":"Khan, ANAS FAHAD ASLAM; Cavallaro, Michele; Cruz Gonz\u00e1lez, Rafael; D\u00edaz-Vera, Javier; Frontini, Francesca; Javier Minaya G\u00f3mez, Francisco","authors_cnr_name":["KHAN, ANAS FAHAD ASLAM","FRONTINI, FRANCESCA"],"authors_cnr_id":["rp05508","rp02790"],"authors_cnr_institute":[],"abstract":"","keywords":[],"pages":"122-124","url":"https:\/\/iris.unive.it\/retrieve\/0f226d38-e332-418b-9b14-d5558d1a0d9d\/AIUCD2023.pdf","volume":"","doi":"","editors":[],"editors_source":"","published":"La Memoria Digitale. Forme Del Testo e Organizzazione Della Conoscenza. Atti Del XII Convegno Annuale AIUCD","publisher":"","issn":"","isbn":"","conference_name":"","conference_place":"","conference_date":"","last_updated_cnr":"2025-01-24 01:49:12","last_updated_oai":"2025-01-24 01:49:12","last_updated_www":"0000-00-00 00:00:00"},{"id":283,"id_source":456225,"institutes":["ILC"],"type":"misc","type_order":11,"title":"Words and the Company they Keep: Digital corpora and infrastructures for the foreign language classroom","year":2023,"authors":["Frontini, F."],"authors_source":"Frontini, Francesca","authors_cnr_name":["FRONTINI, FRANCESCA"],"authors_cnr_id":["rp02790"],"authors_cnr_institute":[],"abstract":"We give an overview of corpora & language technologies and their use in foreign language teaching","keywords":["corpora","didattica L2","tecnologie del linguaggio"],"pages":"","url":"https:\/\/iris.cnr.it\/handle\/20.500.14243\/456225","volume":"","doi":"","editors":[],"editors_source":"","published":"","publisher":"","issn":"","isbn":"","conference_name":"Didattica della lingua, della cultura e cittadinanza attiva: sfide educative contemporanee-Seminari LEND Modena","conference_place":"","conference_date":"","last_updated_cnr":"2025-06-12 01:52:24","last_updated_oai":"2025-06-12 01:52:24","last_updated_www":"0000-00-00 00:00:00"},{"id":74,"id_source":446352,"institutes":["ILC"],"type":"book_chapter","type_order":4,"title":"Language Matters. The European Research Infrastructure CLARIN, Today\u00a0and\u00a0Tomorrow","year":2022,"authors":["De Jong, F.","Van Uytvanck, D.","Frontini, F.","Van Den Bosch, A.","Fi\u0161er, D.","Witt, A."],"authors_source":"de Jong, Franciska; Van Uytvanck, Dieter; Frontini, Francesca; van den Bosch, Antal; Fi\u0161er, Darja; Witt, Andreas","authors_cnr_name":["FRONTINI, FRANCESCA"],"authors_cnr_id":["rp02790"],"authors_cnr_institute":[],"abstract":"LARIN stands for \"Common Language Resources and Technology Infrastructure\". In 2012 CLARIN ERIC was established as a legal entity with the mission to create and maintain a digital infrastructure to support the sharing, use, and sustainability of language data (in written, spoken, or multimodal form) available through repositories from all over Europe, in support of research in the humanities and social sciences and beyond. Since 2016 CLARIN has had the status of Landmark research infrastructure and currently it provides easy and sustainable access to digital language data and also offers advanced tools to discover, explore, exploit, annotate, analyse, or combine such datasets, wherever they are located. This is enabled through a networked federation of centres: language data repositories, service centres, and knowledge centres with single sign-on access for all members of the academic community in all participating countries. In addition, CLARIN offers open access facilities for other interested communities of use, both inside and outside of academia. Tools and data from different centres are interoperable, so that data collections can be combined and tools from different sources can be chained to perform operations at different levels of complexity. The strategic agenda adopted by CLARIN and the activities undertaken are rooted in a strong commitment to the Open Science paradigm and the FAIR data principles. This also enables CLARIN to express its added value for the European Research Area and to act as a key driver of innovation and contributor to the increasing number of industry programmes running on data-driven processes and the digitalization of society at large","keywords":["research infrastructure","language resources","service interoperability","innovation","SSH","language technology","open science"],"pages":"31-58","url":"https:\/\/www.degruyter.com\/document\/doi\/10.1515\/9783110767377-002\/html","volume":"","doi":"10.1515\/9783110767377-002","editors":[],"editors_source":"","published":"CLARIN: The Infrastructure for Language Resources","publisher":"Walter De Gruyter Inc (Boston\/Berlin\/Munich, USA)","issn":"","isbn":"978-3-11-076737-7","conference_name":"","conference_place":"Boston\/Berlin\/Munich","conference_date":"","last_updated_cnr":"2025-03-01 06:43:48","last_updated_oai":"2025-03-01 06:43:48","last_updated_www":"0000-00-00 00:00:00"},{"id":1335,"id_source":419162,"institutes":["ILC"],"type":"book_chapter","type_order":4,"title":"Italian Language Resources. From CLARIN-IT to the VLO and Back: Sketching a Methodology for Monitoring LRs Visibility","year":2022,"authors":["Del Fante, D.","Frontini, F.","Monachini, M.","Quochi, V."],"authors_source":"DEL FANTE, Dario; Frontini, Francesca; Monachini, Monica; Quochi, Valeria","authors_cnr_name":["DEL FANTE, DARIO","FRONTINI, FRANCESCA","MONACHINI, MONICA","QUOCHI, VALERIA"],"authors_cnr_id":["rp14368","rp02790","rp19457","rp13283"],"authors_cnr_institute":[],"abstract":"This paper sketches a user-oriented, qualitative methodology for both (i) monitoring the existence and availability of language resources relevant for a given CLARIN national community and language and (ii) assessing the offering potential of CLARIN, in terms of Language Resources provided to national consortia. From the user perspective, the methodology has been applied to investigate the visibility of language resources available for Italian within the CLARIN central services, in particular the Virtual Language Observatory. As a proof-of-concept, the methodology has been tested on the resources available through the CLARIN-IT data centres, but, ideally, it could be applied by any national data centre aiming to assess the existence of LRs in CLARIN for any given languages and check their accessibility for the interested users. It is thus argued that such an assessment might be a useful instrument in the hands of national coordinators and centre managers for (i) bringing to the fore both strengths and critical issues about their data providing community and (ii) for planning targeted actions to improve and increase both visibility and accessibility of their LRs","keywords":["Virtual Language Observatory","CLARIN-IT","CLARIN-ERIC","Qualitative Assessment Methodology","User Involvement"],"pages":"10-22","url":"https:\/\/ecp.ep.liu.se\/index.php\/clarin\/article\/view\/413\/371","volume":"","doi":"10.3384\/9789179294441","editors":[],"editors_source":"","published":"Selected Papers from the CLARIN Annual Conference 2021","publisher":"","issn":"","isbn":"978-91-7929-444-1","conference_name":"","conference_place":"","conference_date":"","last_updated_cnr":"2025-02-08 06:47:12","last_updated_oai":"2025-02-08 06:47:12","last_updated_www":"0000-00-00 00:00:00"},{"id":963,"id_source":412939,"institutes":["ILC"],"type":"book_chapter","type_order":4,"title":"Approches num\u00e9riques des questions d'auctorialit\u00e9. Le corpus Challe","year":2022,"authors":["Menant","Genevi\u00e8ve","Frontini, F.","Fujiwara","Mami","Martin","Chrostophe"],"authors_source":"Menant, ; Genevi\u00e8ve, ; Frontini, Francesca; Frontini, Francesca; Fujiwara, ; Mami, ; Martin, ; Chrostophe,","authors_cnr_name":["FRONTINI, FRANCESCA","FRONTINI, FRANCESCA"],"authors_cnr_id":["rp02790","rp02790"],"authors_cnr_institute":[],"abstract":"La contribution se concentre sur l'application d'approches textom\u00e9triques et d'identification d'auteur \u00e0 l'oeuvre de Robert Challe","keywords":["Robert Challe","attribution d'auteur","textome\u0301trie"],"pages":"167-192","url":"https:\/\/iris.cnr.it\/handle\/20.500.14243\/412939","volume":"","doi":"10.48611\/isbn.978-2-406-13347-6.p.0167","editors":["Alexandre, D.","Roe, G."],"editors_source":"Alexandre, Didier; Roe, Glenn","published":"Observer la vie litte\u0301raire. E\u0301tudes litte\u0301raires et nume\u0301riques","publisher":"Editions Classiques Garnier (Paris, FRA)","issn":"","isbn":"978-2-406-13347-6","conference_name":"","conference_place":"Paris","conference_date":"","last_updated_cnr":"2025-01-24 23:17:55","last_updated_oai":"2025-01-24 23:17:55","last_updated_www":"0000-00-00 00:00:00"},{"id":248,"id_source":446358,"institutes":["ILC","IGSG"],"type":"conference_article","type_order":7,"title":"Making Italian Parliamentary Records Machine-Actionable: the Construction of the ParlaMint-IT corpus","year":2022,"authors":["Agnoloni, T.","Bartolini, R.","Frontini, F.","Montemagni, S.","Marchetti, C.","Quochi, V.","Ruisi, M.","Venturi, G."],"authors_source":"Agnoloni, Tommaso; Bartolini, Roberto; Frontini, Francesca; Montemagni, Simonetta; Marchetti, Carlo; Quochi, Valeria; Ruisi, Manuela; Venturi, Giulia","authors_cnr_name":["AGNOLONI, TOMMASO","BARTOLINI, ROBERTO","FRONTINI, FRANCESCA","MONTEMAGNI, SIMONETTA","QUOCHI, VALERIA","VENTURI, GIULIA"],"authors_cnr_id":["rp21506","rp00239","rp02790","rp16780","rp13283","rp00732"],"authors_cnr_institute":[],"abstract":"This paper describes the process of acquisition, cleaning, interpretation, coding and linguistic annotation of a collection of parliamentary debates from the Senate of the Italian Republic covering the COVID-19 pandemic emergency period and a former period for reference and comparison according to the CLARIN ParlaMint prescriptions. The corpus contains 1199 sessions and 79, 373 speeches for a total of about 31 million words, and was encoded according to the ParlaCLARIN TEI XML format. It includes extensive metadata about the speakers, sessions, political parties and parliamentary groups. As required by the ParlaMint initiative, the corpus was also linguistically annotated for sentences, tokens, POS tags, lemmas and dependency syntax according to the universal dependencies guidelines. Named entity annotation and classification is also included. All linguistic annotation was performed automatically using state-of-the-art NLP technology with no manual revision. The Italian dataset is freely available as part of the larger ParlaMint 2. 1 corpus deposited and archived in CLARIN repository together with all other national corpora. It is also available for direct analysis and inspection via various CLARIN services and has already been used both for research and educational purposes","keywords":["parliamentary debates","CLARIN ParlaMint","corpus creation","corpus annotation"],"pages":"117-124","url":"https:\/\/aclanthology.org\/2022.parlaclarin-1.17\/","volume":"","doi":"","editors":[],"editors_source":"","published":"Proceedings of The Workshop ParlaCLARIN III within the 13th Language Resources and Evaluation Conference","publisher":"European Language Resources Association ELRA (Paris, FRA)","issn":"","isbn":"979-10-95546-85-6","conference_name":"Workshop ParlaCLARIN III within the 13th Language Resources and Evaluation Conference","conference_place":"Paris","conference_date":"","last_updated_cnr":"2025-05-22 00:35:15","last_updated_oai":"2025-05-22 00:35:15","last_updated_www":"0000-00-00 00:00:00"},{"id":1240,"id_source":416549,"institutes":["ILC"],"type":"conference_article","type_order":7,"title":"CLARIN-IT: An Overview on the Italian Clarin Consortium After Six Years of Activity","year":2022,"authors":["Del Fante, D.","Frontini, F.","Monachini, M.","Quochi, V."],"authors_source":"Dario Del Fante; Francesca Frontini; Monica Monachini; Valeria Quochi","authors_cnr_name":["DEL FANTE, DARIO","FRONTINI, FRANCESCA","MONACHINI, MONICA","QUOCHI, VALERIA"],"authors_cnr_id":["rp14368","rp02790","rp19457","rp13283"],"authors_cnr_institute":[],"abstract":"This paper offers an overview of the Italian CLARIN consortium after six years since its establishment. The members, the centres and the repositories and the most important collections are described. Lastly, in order to showcase the visibility and the accessiblity of Language Resources provided by CLARIN-IT from a user-perspective, we show how Italian resources are findable within CLARIN ERI","keywords":["Language Resources","Data Repositories and Archives","Research Infrastructures","CLARIN"],"pages":"8","url":"http:\/\/ceur-ws.org\/Vol-3160\/short21.pdf","volume":"","doi":"","editors":["Di Nunzio, G. M.","Portelli, B.","Redavid, D.","Silvello, G."],"editors_source":"Giorgio Maria Di Nunzio, Beatrice Portelli, Domenico Redavid, Gianmaria Silvello","published":"Proceedings of the 18th Italian Research Conference on Digital Libraries","publisher":"CEUR-WS. org (Aachen, DEU)","issn":"","isbn":"","conference_name":"Italian Research Conference on Digital Libraries","conference_place":"Aachen","conference_date":"","last_updated_cnr":"2024-06-02 10:50:45","last_updated_oai":"2024-06-02 10:50:45","last_updated_www":"0000-00-00 00:00:00"},{"id":1331,"id_source":446356,"institutes":["ILC"],"type":"conference_article","type_order":7,"title":"Language Technologies for the Creation of Multilingual Terminologies. Lessons Learned from the SSHOC Project","year":2022,"authors":["Gamba, F.","Frontini, F.","Broeder, D.","Monachini, M."],"authors_source":"Gamba, Federica; Frontini, Francesca; Broeder, Daan; Monachini, Monica","authors_cnr_name":["FRONTINI, FRANCESCA","MONACHINI, MONICA"],"authors_cnr_id":["rp02790","rp19457"],"authors_cnr_institute":[],"abstract":"This paper is framed in the context of the SSHOC project and aims at exploring how Language Technologies can help in promoting and facilitating multilingualism in the Social Sciences and Humanities (SSH). Although most SSH researchers produce culturally and societally relevant work in their local languages, metadata and vocabularies used in the SSH domain to describe and index research data are currently mostly in English. We thus investigate Natural Language Processing and Machine Translation approaches in view of providing resources and tools to foster multilingual access and discovery to SSH content across different languages. As case studies, we create and deliver as freely, openly available data a set of multilingual metadata concepts and an automatically extracted multilingual Data Stewardship terminology. The two case studies allow as well to evaluate performances of state-of-the-art tools and to derive a set of recommendations as to how best apply them. Although not adapted to the specific domain, the employed tools prove to be a valid asset to translation tasks. Nonetheless, validation of results by domain experts proficient in the language is an unavoidable phase of the whole workflow","keywords":["language resource infrastructures","Multilingual terminologies","data curation"],"pages":"154-163","url":"https:\/\/aclanthology.org\/2022.lrec-1.17","volume":"","doi":"","editors":[],"editors_source":"","published":"Proceedings of the 13th Language Resources and Evaluation Conference","publisher":"European Language Resources Association ELRA (Paris, FRA)","issn":"","isbn":"979-10-95546-72-6","conference_name":"13th Conference on Language Resources and Evaluation (LREC 2022)","conference_place":"Paris","conference_date":"","last_updated_cnr":"2025-06-14 00:54:40","last_updated_oai":"2025-06-14 00:54:40","last_updated_www":"0000-00-00 00:00:00"},{"id":702,"id_source":419303,"institutes":["ILC"],"type":"conference_article","type_order":7,"title":"Esth\u00e9tique de la voix dans les livres audio en langue fran\u00e7aise","year":2022,"authors":["Hirsch, F.","Frontini, F.","Didirkov\u00e1, I.","Drengubiak, J."],"authors_source":"Fabrice Hirsch; Francesca Frontini; Ivana Didirkov\u00e1;J\u00e1n Drengubiak","authors_cnr_name":["FRONTINI, FRANCESCA"],"authors_cnr_id":["rp02790"],"authors_cnr_institute":[],"abstract":"Cette recherche vise \u00e0 \u00e9tudier les pr\u00e9f\u00e9rences des auditeurs concernant les voix des livres audio. Des \u00e9chantillons de 8 voix masculines et 7 voix f\u00e9minines ont \u00e9t\u00e9 extraits de diff\u00e9rents livres audio et analys\u00e9s. Une enqu\u00eate a \u00e9t\u00e9 r\u00e9alis\u00e9e pour obtenir le point de vue de 69 auditeurs en r\u00e9pondant \u00e0 des questions sur les caract\u00e9ristiques vocales. Les r\u00e9sultats montrent que les choix des participants d\u00e9pendent du genre litt\u00e9raire. En effet, les voix masculines sont pr\u00e9f\u00e9r\u00e9es pour les romans de science-fiction et les voix f\u00e9minines pour la litt\u00e9rature pour enfants et les romans contemporains. N\u00e9anmoins, les autres genres litt\u00e9raires test\u00e9s ne correspondent pas \u00e0 une voix sp\u00e9cifique. Concernant le d\u00e9bit, une pr\u00e9f\u00e9rence a \u00e9t\u00e9 not\u00e9e pour des essais lus avec un d\u00e9bit de parole plus lent, alors que les auditeurs pr\u00e9f\u00e8rent un d\u00e9bit de parole plus rapide pour les romans \u00e9rotiques","keywords":["audiobooks","voice esthetics","speech"],"pages":"","url":"https:\/\/doi.org\/10.1051\/shsconf\/202213808004","volume":"","doi":"10.1051\/shsconf\/202213808004","editors":[],"editors_source":"","published":"138","publisher":"","issn":"","isbn":"","conference_name":"8e Congre\u0300s Mondial de Linguistique Franc\u0327aise","conference_place":"","conference_date":"","last_updated_cnr":"2024-03-23 00:21:52","last_updated_oai":"2024-03-23 00:21:52","last_updated_www":"0000-00-00 00:00:00"},{"id":209,"id_source":412365,"institutes":["ILC"],"type":"technical_report","type_order":9,"title":"D3. 8 Lexical-semantic analytics for NLP","year":2022,"authors":["Martelli, F.","Maru, M.","Campagnano, C.","Navigli, R.","Velardi, P.","Ure\u00f1aruiz, R.","Frontini, F.","Quochi, V.","Kallas, J.","Koppel, K.","Langemets, M.","De Does, J.","Tempelaars, R.","Tiberius, C.","Costa, R.","Salgado, A.","Krek, S.","Ibej, J.","Dobrovoljc, K.","Gantar, P.","Munda, T."],"authors_source":"Martelli, Federico; Maru, Marco; Campagnano, Cesare; Navigli, Roberto; Velardi, Paola; Ure\u00f1aruiz, Rafaelj; Frontini, Francesca; Quochi, Valeria; Kallas, Jelena; Koppel, Kristina; Langemets, Margit; de Does, Jesse; Tempelaars, Rob; Tiberius, Carole; Costa, Rute; Salgado, Ana; Krek, Simon; Ibej, Jaka; Dobrovoljc, Kaja; Gantar, Polona; Munda, Tina","authors_cnr_name":["FRONTINI, FRANCESCA","QUOCHI, VALERIA"],"authors_cnr_id":["rp02790","rp13283"],"authors_cnr_institute":[],"abstract":"The present document illustrates the work carried out in task 3. 3 (work package 3) focused on lexicalsemantic analytics for Natural Language Processing (NLP). This task aims at computing analytics for lexicalsemantic information such as words, senses and domains in the available resources, investigating their role in NLP applications. Specifically, this task concentrates on three research directions, namely i) which grouping senses based on their semantic similari sense clustering, in ty improves the performance of NLP tasks such as Word Sense Disambiguation (WSD), ii) domain labeling of text, in which the lexicographic resources made available by the ELEXIS project for research purposes allow better performances to be achieved, and fin senses ally iii) analysing the, for which a software package is made available. diachronic distribution of In this deliverable, we illustrate the research activities aimed at achieving the aforementioned goals and put forward suggestions for future works. Importantly, we stress the crucial role played by highquality lexicalsemantic r esources when investigating such linguistic aspects and their impact on NLP applications. To this end, as an additional contribution, we address the paucity of manually the ELEXIS parallelannotated data in the lexical senseannotated datasetsemantic research field and introduce, a novel entirely manuallyavailable in 10 European languages and featuring 5 annotation layers","keywords":["research infrastructures","lexicography","lexical resources","word-sense disambiguation","WSD","sense-annotated language data","multilinguality"],"pages":"67","url":"https:\/\/elex.is\/wp-content\/uploads\/ELEXIS_D3_8_Lexical-Semantic_Analytics_for_NLP_final_report.pdf","volume":"","doi":"","editors":[],"editors_source":"","published":"","publisher":"","issn":"","isbn":"","conference_name":"","conference_place":"","conference_date":"","last_updated_cnr":"2024-05-12 00:21:36","last_updated_oai":"2024-05-12 00:21:36","last_updated_www":"0000-00-00 00:00:00"},{"id":171,"id_source":446092,"institutes":["ILC"],"type":"technical_report","type_order":9,"title":"D5. 3 Overview of Online Tutorials and Instruction Manuals","year":2022,"authors":["Tasovac, T.","Tiberius, C.","Bamberg, C.","Bellandi, A.","Burch, T.","Costa, R.","Uro, M.","Frontini, F.","Hennemann, J.","Heylen, K.","Milojakub\u00edek","Khan, F.","Klee, A.","Kosem, I.","Kov\u00e1, V.","Matuka, O.","McCrae, J.","Monachini, M.","M\u00f6rth, K.","Munda, T.","Quochi, V.","Andrarepar","Roche, C.","Salgado, A.","Sievers, H.","V\u00e1radi, T.","Weyand, S.","Woldrich, A.","Zhanial, S."],"authors_source":"Tasovac, Toma; Tiberius, Carole; Bamberg, Claudia; Bellandi, Andrea; Burch, Thomas; Costa, Rute; Uro, Matej; Frontini, Francesca; Hennemann, Julia; Heylen, Kris; Milojakub\u00edek, ; Khan, Fahad; Klee, Anne; Kosem, Iztok; Kov\u00e1, Vojtch; Matuka, Ondej; Mccrae, John; Monachini, Monica; M\u00f6rth, Karlheinz; Munda, Tina; Quochi, Valeria; Andrarepar, ; Roche, Christophe; Salgado, Ana; Sievers, Henrike; V\u00e1radi, Tam\u00e1s; Weyand, Sandra; Woldrich, Anna; Zhanial, Susanne","authors_cnr_name":["BELLANDI, ANDREA","FRONTINI, FRANCESCA","MONACHINI, MONICA","QUOCHI, VALERIA"],"authors_cnr_id":["rp05868","rp02790","rp19457","rp13283"],"authors_cnr_institute":[],"abstract":"The ELEXIS Curriculum is an integrated set of training materials which contextualizes ELEXIS tools and services inside a broader, systematic pedagogic narrative. This means that the goal of the ELEXIS Curriculum is not simply to inform users about the functionalities of particular tools and services developed within the project, but to show how such tools and services are a) embedded in both lexicographic theory and practice; and b) representative of and contributing to the development of digital skills among lexicographers. The scope and rationale of the curriculum are described in more detail in the Deliverable D5. 2 Guidelines for Producing ELEXIS Tutorials and Instruction Manuals. The goal of this deliverable, as stated in the project DOW, is to provide \"a clear, structured overview of tutorials and instruction manuals developed within the project. \"","keywords":["ELEXIS","lexicography","training materials"],"pages":"31","url":"https:\/\/elex.is\/wp-content\/uploads\/ELEXIS_D5_3_Overview-of-Online-Tutorials-and-Instruction-Manuals.pdf","volume":"","doi":"","editors":[],"editors_source":"","published":"","publisher":"","issn":"","isbn":"","conference_name":"","conference_place":"","conference_date":"","last_updated_cnr":"2024-04-24 11:35:41","last_updated_oai":"2024-04-24 11:35:41","last_updated_www":"0000-00-00 00:00:00"},{"id":276,"id_source":441101,"institutes":["ILC"],"type":"misc","type_order":11,"title":"CLARIN Tools and Resources for Lexicographic Work","year":2022,"authors":["Frontini, F.","Bellandi, A.","Quochi, V.","Monachini, M.","M\u00f6rth, K.","Zhanial, S.","\u010eur\u010do, M.","Woldrich, A."],"authors_source":"Francesca FrontiniAndrea BellandiValeria QuochiMonica MonachiniKarlheinz M\u00f6rthSusanne ZhanialMatej uroAnna Woldrich","authors_cnr_name":["QUOCHI, VALERIA"],"authors_cnr_id":["rp13283"],"authors_cnr_institute":[],"abstract":"This course introduces lexicographers to the CLARIN Research Infrastructure and highlights language resources and tools useful for lexicographic practices. The course consists of two parts. In Part 1, you will learn about CLARIN, its technical and knowledge infrastructure, and about how to deposit and find lexical resources in CLARIN. In Part 2, you will become acquainted with CLARIN tools that can be used to create lexical resources","keywords":["CLARIN","lexicography"],"pages":"","url":"https:\/\/elexis.humanistika.org\/id\/UnwYPq70Dewbn7XDEjsMM","volume":"","doi":"","editors":[],"editors_source":"","published":"","publisher":"","issn":"","isbn":"","conference_name":"","conference_place":"","conference_date":"","last_updated_cnr":"2025-03-03 11:02:45","last_updated_oai":"2025-03-03 11:02:45","last_updated_www":"0000-00-00 00:00:00"},{"id":1342,"id_source":446359,"institutes":["ILC"],"type":"misc","type_order":11,"title":"Parallel sense-annotated corpus ELEXIS-WSD 1. 0","year":2022,"authors":["Martelli, F.","Navigli, R.","Krek, S.","Kallas, J.","Gantar, P.","Koeva, S.","Nimb, S.","Sandford Pedersen, B.","Olsen, S.","Langemets, M.","Koppel, K.","\u00dcksik, T.","Dobrovoljc, K.","Ure\u00f1aruiz, R.","Sanchos\u00e1nchez, J.","Lipp, V.","V\u00e1radi, T.","Gyrffy, A.","L\u00e1szl\u00f3, S.","Quochi, V.","Monachini, M.","Frontini, F.","Tiberius, C.","Tempelaars, R.","Costa, R.","Salgado, A.","Ibej, J.","Munda, T."],"authors_source":"Martelli, Federico; Navigli, Roberto; Krek, Simon; Kallas, Jelena; Gantar, Polona; Koeva, Svetla; Nimb, Sanni; Sandford Pedersen, Bolette; Olsen, Sussi; Langemets, Margit; Koppel, Kristina; \u00dcksik, Tiiu; Dobrovoljc, Kaja; Ure\u00f1aruiz, Rafael; Sanchos\u00e1nchez, Jos\u00e9luis; Lipp, Veronika; V\u00e1radi, Tam\u00e1s; Gyrffy, Andr\u00e1s; L\u00e1szl\u00f3, Simon; Quochi, Valeria; Monachini, Monica; Frontini, Francesca; Tiberius, Carole; Tempelaars, Rob; Costa, Rute; Salgado, Ana; Ibej, Jaka; Munda, Tina","authors_cnr_name":["QUOCHI, VALERIA","MONACHINI, MONICA","FRONTINI, FRANCESCA"],"authors_cnr_id":["rp13283","rp19457","rp02790"],"authors_cnr_institute":[],"abstract":"ELEXIS-WSD is a parallel sense-annotated corpus in which content words (nouns, adjectives, verbs, and adverbs) have been assigned senses. Version 1. 0 contains sentences for 10 languages: Bulgarian, Danish, English, Spanish, Estonian, Hungarian, Italian, Dutch, Portuguese, and Slovene. The corpus was compiled by automatically extracting a set of sentences from WikiMatrix (Schwenk et al., 2019), a large open-access collection of parallel sentences derived from Wikipedia, using an automatic approach based on multilingual sentence embeddings. The sentences were manually validated according to specific formal, lexical and semantic criteria (e. g. by removing incorrect punctuation, morphological errors, notes in square brackets and etymological information typically provided in Wikipedia pages). To obtain a satisfying semantic coverage, we filtered out sentences with less than 5 words and less than 2 polysemous words were filtered out. Subsequently, in order to obtain datasets in the other nine target languages, for each selected sentence in English, the corresponding WikiMatrix translation into each of the other languages was retrieved. If no translation was available, the English sentence was translated manually. The resulting corpus is comprised of 2, 024 sentences for each language","keywords":["Word Sense Disambiguation","corpus parallelo","disambiguazione automatica del senso","annotazione semantica multilingue"],"pages":"","url":"https:\/\/iris.cnr.it\/handle\/20.500.14243\/446359","volume":"","doi":"","editors":[],"editors_source":"","published":"","publisher":"","issn":"","isbn":"","conference_name":"","conference_place":"","conference_date":"","last_updated_cnr":"2025-03-06 00:30:50","last_updated_oai":"2025-03-06 00:30:50","last_updated_www":"0000-00-00 00:00:00"},{"id":833,"id_source":394922,"institutes":["ILC"],"type":"book_chapter","type_order":4,"title":"WEIR-P: An Information Extraction Pipeline for the Wastewater Domain","year":2021,"authors":["Chahinian, N.","Bonnabaud La Bruy\u00e8re, T.","Frontini, F.","Delenne, C.","Julien, M.","Panckhurst, R.","Roche, M.","Sautot, L.","Deruelle, L.","Teisseire, M."],"authors_source":"Chahinian, Nan\u00e9e; Bonnabaud La Bruy\u00e8re, Thierry; Frontini, Francesca; Delenne, Carole; Julien, Marin; Panckhurst, Rachel; Roche, Mathieu; Sautot, Lucile; Deruelle, Laurent; Teisseire, Maguelonne","authors_cnr_name":["FRONTINI, FRANCESCA"],"authors_cnr_id":["rp02790"],"authors_cnr_institute":[],"abstract":"We present the MeDO project, aimed at developing resources for text mining and information extraction in the wastewater domain. We developed a specific Natural Language Processing (NLP) pipeline named WEIR-P (WastewatEr InfoRmation extraction Platform) which identifies the entities and relations to be extracted from texts, pertaining to information, wastewater treatment, accidents and works, organizations, spatio-temporal information, measures and water quality. We presentand evaluate the first version of the NLP system which was developed to automate the extraction of the aforementioned annotation from texts and its integration with existing domain knowledge. The preliminary results obtained on the Montpellier corpus are encouraging and show how a mix of supervised and rule-based techniques can be used to extract useful information and reconstruct the various phases of the extension of a given wastewater network. While the NLP and Information Extraction (IE) methods used are state of the art, the novelty of our work lies in their adaptation to the domain, and in particular in the wastewater management conceptual model, which defines the relations between entities. French resources are less developed in the NLP community than English ones. The datasets obtained in this project are another original aspect of this work","keywords":["Wastewater","text mining","Information extraction","NLP","NER","Domain adapted systems"],"pages":"171-188","url":"https:\/\/www.springer.com\/gp\/book\/9783030750176","volume":"","doi":"","editors":[],"editors_source":"","published":"Research Challenges in Information Science-15th International Conference, RCIS 2021, Limassol, Cyprus, May 11-14, 2021, Proceedings","publisher":"Springer Nature Switzerland (Basel, CHE)","issn":"","isbn":"978-3-030-75017-6","conference_name":"","conference_place":"Basel","conference_date":"","last_updated_cnr":"2025-03-06 00:18:33","last_updated_oai":"2025-03-06 00:18:33","last_updated_www":"0000-00-00 00:00:00"},{"id":1196,"id_source":397005,"institutes":["ILC"],"type":"book_chapter","type_order":4,"title":"An Internationally Fair Mediated Digital Discourse Corpus: Improving Knowledge on Reuse","year":2021,"authors":["Panckhurst, R.","Frontini, F."],"authors_source":"Panckhurst, Rachel; Frontini, Francesca","authors_cnr_name":["FRONTINI, FRANCESCA"],"authors_cnr_id":["rp02790"],"authors_cnr_institute":[],"abstract":"In this paper, the authors present a French Mediated Digital Discourse corpus, (88milSMShttp: \/\/88milsms. huma-num. fr https: \/\/hdl. handle. net\/11403\/comere\/cmr-88milsms). Efforts were undertaken over the years to ensure its publication accordingto the best practices and standards of the community, thus guaranteeing compliance with FAIRprinciples and CLARIN recommendations with pertinent scientific and pedagogical reuse. Sinceknowledge on how resources are reused is sometimes difficult to obtain, ways of improving thisare also envisaged","keywords":["Reuse","FAIR","SMS","corpus"],"pages":"185-193","url":"https:\/\/ecp.ep.liu.se\/index.php\/clarin\/article\/view\/20","volume":"180","doi":"10.3384\/ecp18020","editors":[],"editors_source":"","published":"Selected Papers from the CLARIN Annual Conference 2020","publisher":"","issn":"","isbn":"","conference_name":"","conference_place":"","conference_date":"","last_updated_cnr":"2025-02-05 22:56:41","last_updated_oai":"2025-02-05 22:56:41","last_updated_www":"0000-00-00 00:00:00"},{"id":1854,"id_source":447069,"institutes":["ILC"],"type":"conference_article","type_order":7,"title":"CLARIN-IT Resources in CLARIN ERIC-a Bird's-Eye View","year":2021,"authors":["Del Fante, D.","Frontini, F.","Monachini, M.","Quochi, V."],"authors_source":"DEL FANTE, Dario; Frontini, Francesca; Monachini, Monica; Quochi, Valeria","authors_cnr_name":["DEL FANTE, DARIO","FRONTINI, FRANCESCA","MONACHINI, MONICA","QUOCHI, VALERIA"],"authors_cnr_id":["rp14368","rp02790","rp19457","rp13283"],"authors_cnr_institute":[],"abstract":"The paper investigates the visibility of CLARIN-IT language resources within the services of the CLARINERICcentral infrastructure, notably the Virtual Language Observatory, the Switchboard and the Federated Content Search, from a user perspective in order to identify possible issues. While the experiment focused on one national consortium, the ultimate goal is to develop an assessment methodology that can be used by any national consortia aiming to review the accessibility of their resources and tools within the CLARIN central services","keywords":["FAIR","research infrastructure for SSH","language resources","findability","CLARIN"],"pages":"129-133","url":"https:\/\/office.clarin.eu\/v\/CE-2021-1923-CLARIN2021_ConferenceProceedings.pdf","volume":"","doi":"","editors":[],"editors_source":"","published":"","publisher":"","issn":"2021-1923","isbn":"","conference_name":"CLARIN Annual Conference 2021","conference_place":"","conference_date":"","last_updated_cnr":"2024-04-20 12:50:11","last_updated_oai":"2024-04-20 12:50:11","last_updated_www":"0000-00-00 00:00:00"},{"id":222,"id_source":443238,"institutes":["ILC"],"type":"conference_article","type_order":7,"title":"Designing the ELEXIS Parallel Sense-Annotated Dataset in 10 European Languages","year":2021,"authors":["Martelli, F.","Navigli, R.","Krek, S.","Tiberius, C.","Kallas, J.","Gantar, P.","Koeva, S.","Nimb, S.","Pedersen, B. S.","Olsen, S.","Langements, M.","Koppel, K.","\u00dcksik, T.","Dobrovolijc, K.","Ure\u00f1a Ruiz, R. J.","Sancho S\u00e1nchez, J. L.","Lipp, V.","V\u00e1radi, T.","Gy\u0151rffy, A.","L\u00e1szl\u00f3, S.","Quochi, V.","Monachini, M.","Frontini, F.","Tempelaars, R.","Costa, R.","Salgado, A.","\u010cibej, J.","Munda, T."],"authors_source":"Martelli, ; Federico, ; Navigli, ; Roberto, ; Krek, ; Simon, ; Tiberius, ; Carole, ; Kallas, ; Jelena, ; Gantar, ; Polona, ; Koeva, ; Svetla, ; Nimb, ; Sanni, ; Pedersen, ; Bolette, Sandford; Olsen, ; Sussi, ; Langements, ; Margit, ; Koppel, ; Kristina, ; Ksik, ; Tiiu, ; Dobrovolijc, ; Kaja, ; Urearuiz, ; Rafaelj, ; Sanchosnchez, ; Josluis, ; Lipp, ; Veronika, ; Varadi, ; Tamas, ; Gyrffy, ; Andrs, ; Lszl, ; Simon, ; Quochi, Valeria; Quochi, Valeria; Monachini, Monica; Monachini, Monica; Frontini, Francesca; Frontini, Francesca; Tempelaars, ; Rob, ; Costa, ; Rute, ; Salgado, ; Ana, ; Ibej, ; Jaka, ; Munda, ; Tina,","authors_cnr_name":["QUOCHI, VALERIA","QUOCHI, VALERIA","MONACHINI, MONICA","MONACHINI, MONICA","FRONTINI, FRANCESCA","FRONTINI, FRANCESCA"],"authors_cnr_id":["rp13283","rp13283","rp19457","rp19457","rp02790","rp02790"],"authors_cnr_institute":[],"abstract":"Over the course of the last few years, lexicography has witnessed the burgeoning of increasingly reliable automaticapproaches supporting the creation of lexicographic resources such as dictionaries, lexical knowledge bases andannotated datasets. In fact, recent achievements in the field of Natural Language Processing and particularly inWord Sense Disambiguation have widely demonstrated their effectiveness not only for the creation of lexicographicresources, but also for enabling a deeper analysis of lexical-semantic data both within and across languages. Nevertheless, we argue that the potential derived from the connections between the two fields is far from exhausted. In this work, we address a serious limitation affecting both lexicography and Word Sense Disambiguation, i. e. thelack of high-quality sense-annotated data and describe our efforts aimed at constructing a novel entirely manuallyannotated parallel dataset in 10 European languages. For the purposes of the present paper, we concentrate on theannotation of morpho-syntactic features. Finally, unlike many of the currently available sense-annotated datasets, we will annotate semantically by using senses derived from high-quality lexicographic repositories","keywords":["Digital lexicography","Word Sense Disambiguation","Computational Linguistics","Corpus Linguistics","Natural Language Processing"],"pages":"377-395","url":"https:\/\/static-curis.ku.dk\/portal\/files\/279888836\/eLex_2021_22_pp377_395.pdf","volume":"2021","doi":"","editors":[],"editors_source":"","published":"Electronic lexicography in the 21st century (eLex 2021): Post-editing lexicography","publisher":"Lexical Computing (Brno, CZE)","issn":"","isbn":"","conference_name":"eLex 2021","conference_place":"Brno","conference_date":"","last_updated_cnr":"2025-03-05 23:36:44","last_updated_oai":"2025-03-05 23:36:44","last_updated_www":"0000-00-00 00:00:00"},{"id":1650,"id_source":394923,"institutes":["ILC"],"type":"technical_report","type_order":9,"title":"Guide d'annotation manuelle d'entit\u00e9s nomm\u00e9es dans des corpus litt\u00e9raires","year":2021,"authors":["Alrahabi, M.","Brando, C.","Frontini, F.","Provenier, A.","Jalabert, R.","Bordry, M.","Koskas, C.","Gawley, J."],"authors_source":"Alrahabi, Motasem; Brando, Carmen; Frontini, Francesca; Provenier, Arthur; Jalabert, Romain; Bordry, Margarite; Koskas, Camille; Gawley, James","authors_cnr_name":["FRONTINI, FRANCESCA"],"authors_cnr_id":["rp02790"],"authors_cnr_institute":[],"abstract":"Guide d'annotation manuelle d'entit\u00e9s nomm\u00e9es dans des corpus litt\u00e9raires Campagne d'annotation OBVIL 2019-2021","keywords":["NER"],"pages":"","url":"https:\/\/hal.archives-ouvertes.fr\/hal-03156278","volume":"","doi":"","editors":[],"editors_source":"","published":"","publisher":"","issn":"","isbn":"","conference_name":"","conference_place":"","conference_date":"","last_updated_cnr":"2024-06-14 13:11:56","last_updated_oai":"2024-06-14 13:11:56","last_updated_www":"0000-00-00 00:00:00"},{"id":2175,"id_source":446080,"institutes":["IGSG","ILC"],"type":"misc","type_order":11,"title":"Multilingual comparable corpora of parliamentary debates ParlaMint 2. 1","year":2021,"authors":["Erjavec, T.","Ogrodniczuk, M.","Osenova, P.","Ljubei, N.","Simov, K.","Grigorova, V.","Rudolf, M.","Panur, A.","Kopp, M.","Barkarson, S.","Steingr\u00edmsson, S.","Van Der Pol, H.","Depoorter, G.","De Does, J.","Jongejan, B.","Haltrup Hansen, D.","Navarretta, C.","Calzada P\u00e9rez, M.","D De Macedo, L.","Van Heusden, R.","Marx, M.","\u00c7\u00f6ltekin, \u00c7.","Coole, M.","Agnoloni, T.","Frontini, F.","Montemagni, S.","Quochi, V.","Venturi, G.","Ruisi, M.","Marchetti, C.","Battistoni, R.","Sebk, M.","Ring, O.","Daris, R.","Utka, A.","Petkeviius, M.","Briedien\u00e9, M.","Krilaviius, T.","Morkeviius, V.","Bartolini, R.","Cimino, A.","Diwersy, S.","Luxardo, G.","Rayson, P."],"authors_source":"Erjavec, Toma; Ogrodniczuk, Maciej; Osenova, Petya; Ljubei, Nikola; Simov, Kiril; Grigorova, Vladislava; Rudolf, Micha; Panur, Andrej; Kopp, Maty\u00e1; Barkarson, Starka\u00f0ur; Steingr\u00edmsson, Stein\u00feor; van der Pol, Henk; Depoorter, Griet; de Does, Jesse; Jongejan, Bart; Haltrup Hansen, Dorte; Navarretta, Costanza; Calzada P\u00e9rez, Mar\u00eda; D de Macedo, Luciana; van Heusden, Ruben; Marx, Maarten; \u00c7\u00f6ltekin, \u00c7ar; Coole, Matthew; Agnoloni, Tommaso; Frontini, Francesca; Montemagni, Simonetta; Quochi, Valeria; Venturi, Giulia; Ruisi, Manuela; Marchetti, Carlo; Battistoni, Roberto; Sebk, Mikl\u00f3s; Ring, Orsolya; Daris, Roberts; Utka, Andrius; Petkeviius, Mindaugas; Briedien\u00e9, Monika; Krilaviius, Tomas; Morkeviius, Vaidas; Bartolini, Roberto; Cimino, Andrea; Diwersy, Sascha; Luxardo, Giancarlo; Rayson, Paul","authors_cnr_name":["AGNOLONI, TOMMASO","FRONTINI, FRANCESCA","MONTEMAGNI, SIMONETTA","QUOCHI, VALERIA","VENTURI, GIULIA","BARTOLINI, ROBERTO"],"authors_cnr_id":["rp21506","rp02790","rp16780","rp13283","rp00732","rp00239"],"authors_cnr_institute":[],"abstract":"ParlaMint 2. 1 is a multilingual set of 17 comparable corpora containing parliamentary debates mostly starting in 2015 and extending to mid-2020, with each corpus being about 20 million words in size. The sessions in the corpora are marked as belonging to the COVID-19 period (after November 1st 2019), or being \"reference\" (before that date). The corpora have extensive metadata, including aspects of the parliament; the speakers (name, gender, MP status, party affiliation, party coalition\/opposition); are structured into time-stamped terms, sessions and meetings; with speeches being marked by the speaker and their role (e. g. chair, regular speaker). The speeches also contain marked-up transcriber comments, such as gaps in the transcription, interruptions, applause, etc. Note that some corpora have further information, e. g. the year of birth of the speakers, links to their Wikipedia articles, their membership in various committees, etc. The corpora are encoded according to the Parla-CLARIN TEI recommendation (https: \/\/clarin-eric. github. io\/parla-clarin\/), but have been validated against the compatible, but much stricter ParlaMint schemas. This entry contains the ParlaMint TEI-encoded corpora with the derived plain text version of the corpus along with TSV metadata on the speeches. Also included is the 2. 0 release of the data and scripts available at the GitHub repository of the ParlaMint project. Note that there also exists the linguistically marked-up version of the corpus, which is available at http: \/\/hdl. handle. net\/11356\/1431","keywords":[],"pages":"","url":"https:\/\/iris.cnr.it\/handle\/20.500.14243\/446080","volume":"","doi":"","editors":[],"editors_source":"","published":"","publisher":"","issn":"","isbn":"","conference_name":"","conference_place":"","conference_date":"","last_updated_cnr":"2025-03-07 06:32:49","last_updated_oai":"0000-00-00 00:00:00","last_updated_www":"0000-00-00 00:00:00"},{"id":383,"id_source":446076,"institutes":["IGSG","ILC"],"type":"misc","type_order":11,"title":"Linguistically annotated multilingual comparable corpora of parliamentary debates ParlaMint. ana 2. 1","year":2021,"authors":["Erjavec, T.","Ogrodniczuk, M.","Osenova, P.","Ljubei, N.","Simov, K.","Grigorova, V.","Rudolf, M.","Panur, A.","Kopp, M.","Barkarson, S.","Steingr\u00edmsson, S.","Van Der Pol, H.","Depoorter, G.","De Does, J.","Jongejan, B.","Haltrup Hansen, D.","Navarretta, C.","Calzada P\u00e9rez, M.","D De Macedo, L.","Van Heusden, R.","Marx, M.","\u00c7\u00f6ltekin, \u00c7.","Coole, M.","Agnoloni, T.","Frontini, F.","Montemagni, S.","Quochi, V.","Venturi, G.","Ruisi, M.","Marchetti, C.","Battistoni, R.","Sebk, M.","Ring, O.","Daris, R.","Utka, A.","Petkeviius, M.","Briedien\u00e9, M.","Krilaviius, T.","Morkeviius, V.","Bartolini, R.","Cimino, A.","Diwersy, S.","Luxardo, G.","Rayson, P."],"authors_source":"Erjavec, Toma; Ogrodniczuk, Maciej; Osenova, Petya; Ljubei, Nikola; Simov, Kiril; Grigorova, Vladislava; Rudolf, Micha; Panur, Andrej; Kopp, Maty\u00e1; Barkarson, Starka\u00f0ur; Steingr\u00edmsson, Stein\u00feor; van der Pol, Henk; Depoorter, Griet; de Does, Jesse; Jongejan, Bart; Haltrup Hansen, Dorte; Navarretta, Costanza; Calzada P\u00e9rez, Mar\u00eda; D de Macedo, Luciana; van Heusden, Ruben; Marx, Maarten; \u00c7\u00f6ltekin, \u00c7ar; Coole, Matthew; Agnoloni, Tommaso; Frontini, Francesca; Montemagni, Simonetta; Quochi, Valeria; Venturi, Giulia; Ruisi, Manuela; Marchetti, Carlo; Battistoni, Roberto; Sebk, Mikl\u00f3s; Ring, Orsolya; Daris, Roberts; Utka, Andrius; Petkeviius, Mindaugas; Briedien\u00e9, Monika; Krilaviius, Tomas; Morkeviius, Vaidas; Bartolini, Roberto; Cimino, Andrea; Diwersy, Sascha; Luxardo, Giancarlo; Rayson, Paul","authors_cnr_name":["AGNOLONI, TOMMASO","FRONTINI, FRANCESCA","MONTEMAGNI, SIMONETTA","QUOCHI, VALERIA","VENTURI, GIULIA","BARTOLINI, ROBERTO","CIMINO, ANDREA"],"authors_cnr_id":["rp21506","rp02790","rp16780","rp13283","rp00732","rp00239","rp05770"],"authors_cnr_institute":[],"abstract":"ParlaMint 2. 1 is a multilingual set of 17 comparable corpora containing parliamentary debates mostly starting in 2015 and extending to mid-2020, with each corpus being about 20 million words in size. The sessions in the corpora are marked as belonging to the COVID-19 period (from November 1st 2019), or being \"reference\" (before that date). The corpora have extensive metadata, including aspects of the parliament; the speakers (name, gender, MP status, party affiliation, party coalition\/opposition); are structured into time-stamped terms, sessions and meetings; with speeches being marked by the speaker and their role (e. g. chair, regular speaker). The speeches also contain marked-up transcriber comments, such as gaps in the transcription, interruptions, applause, etc. Note that some corpora have further information, e. g. the year of birth of the speakers, links to their Wikipedia articles, their membership in various committees, etc. The corpora are encoded according to the Parla-CLARIN TEI recommendation (https: \/\/clarin-eric. github. io\/parla-clarin\/), but have been validated against the compatible, but much stricter ParlaMint schemas. This entry contains the linguistically marked-up version of the corpus, while the text version is available at http: \/\/hdl. handle. net\/11356\/1432. The ParlaMint. ana linguistic annotation includes tokenization, sentence segmentation, lemmatisation, Universal Dependencies part-of-speech, morphological features, and syntactic dependencies, and the 4-class CoNLL-2003 named entities. Some corpora also have further linguistic annotations, such as PoS tagging or named entities according to language-specific schemes, with their corpus TEI headers giving further details on the annotation vocabularies and tools","keywords":["covid-19","ParlaCLARIN","CLARIN","linguistic annotation","pos-tagging","Named Entity Recognition","linguistic dependency annotation","UD","dibattiti parlamentari","parlamenti","discorso politico"],"pages":"","url":"https:\/\/iris.cnr.it\/handle\/20.500.14243\/446076","volume":"","doi":"","editors":[],"editors_source":"","published":"","publisher":"","issn":"","isbn":"","conference_name":"","conference_place":"","conference_date":"","last_updated_cnr":"2025-03-07 22:39:26","last_updated_oai":"2025-03-07 22:39:26","last_updated_www":"0000-00-00 00:00:00"},{"id":1443,"id_source":529601,"institutes":["ILC"],"type":"journal_article","type_order":1,"title":"Pratiques de gestion des donn\u00e9es de la\u00a0recherche\u00a0: une\u00a0n\u00e9cessaire acculturation des\u00a0chercheurs aux\u00a0enjeux de\u00a0la\u00a0science\u00a0ouverte\u00a0?","year":2020,"authors":["Amiel, P.","Frontini, F.","Lacour, P. Y.","Robin, A."],"authors_source":"Amiel, Philippe; Frontini, Francesca; Lacour, Pierre-Yves; Robin, Agn\u00e8s","authors_cnr_name":["FRONTINI, FRANCESCA"],"authors_cnr_id":["rp02790"],"authors_cnr_institute":[],"abstract":"The article presents the results of an exploratory survey, conducted in June 2018 in the Montpellier\u2019s basin by the CommonData Research Program, on the researchers\u2019 management practices of research data. The principles objectives were to see if research data management is the result of an elaborated and strategic plan, to verify the ability or inability of researchers to qualify legally their explored, collected or produced datasets in order to determine their management in regard of the current Open Science politics and, at last, to observe the property feeling improved by researchers toward the datas they contribute to produce, which comes up with the broader question of the personal and\/or institutional dimension of research\u2019s work and its consequences in the awarding of property","keywords":["research data, management, property, sharing, dissemination, valorization, public domain, open science","donne\u0301es de la recherche, gestion, proprie\u0301te\u0301, partage, diffusion, valorisation, domaine public, science ouverte"],"pages":"147-168","url":"https:\/\/iris.cnr.it\/handle\/20.500.14243\/529601","volume":"(10)","doi":"10.4000\/cdst.2061","editors":[],"editors_source":"","published":"CAHIERS DROIT, SCIENCES & TECHNOLOGIES","publisher":"","issn":"1967-0311","isbn":"","conference_name":"","conference_place":"","conference_date":"","last_updated_cnr":"2025-02-05 23:07:39","last_updated_oai":"2025-02-05 23:07:39","last_updated_www":"0000-00-00 00:00:00"},{"id":827,"id_source":529606,"institutes":["ILC"],"type":"book_chapter","type_order":4,"title":"Evolving interactional practices of emoji in text messages","year":2020,"authors":["Panckhurst, R.","Frontini, F."],"authors_source":"Panckhurst, Rachel; Frontini, Francesca","authors_cnr_name":["FRONTINI, FRANCESCA"],"authors_cnr_id":["rp02790"],"authors_cnr_institute":[],"abstract":"In this article, we examine the usage of emoji in the 88milSMS corpus. After differentiating between emoji and emoticons, we situate the context, indicate general statistics and mention press interest. Next, we address linguistic issues: are emoji used more often in addition (either redundantly or necessarily, sometimes as \u201csofteners\u201d (adoucisseurs, D\u00e9trie &amp; Verine 2015) or for lexical replacement, denoting a reference\/referential function (Referenzfunktion, D\u00fcrscheid &amp; Siever 2017)? Concerning emoji insertion positioning, which is the most popular and what does this mean? Other researchers refer to \u201cthe emoji code\u201d (Danesi 2016; Evans 2017), and emoji classifications have been proposed, including references to syntactic, semantic (Barbieri, Ronzano &amp; Saggion 2016), semiotic, phatic and emotive\/sentiment (Novak et al. 2015) levels. Are these satisfactory or do we need to redefine levels, contexts and potential ambiguity? Part-ofspeech tagging (POS) and NLP software are then used to annotate SMS containing emoji within 88milSMS in order to investigate the immediate grammatical environment. This allows us to conduct contextual analysis relating to syntactic linguistic functions of emoji. Finally, results from two questionnaires are explored: 1. sociolinguistic factors (age, gender) of the SMS donors having used emoji in 88milSMS; 2. Comparison of SMS emoji usage with other instant messaging applications and social networks via a user-orientated questionnaire (Rascol 20171)","keywords":["emoji, computer mediated communication, corpus"],"pages":"81-104","url":"https:\/\/doi.org\/10.1515\/9781501510113-005","volume":"","doi":"10.1515\/9781501510113-005","editors":[],"editors_source":"","published":"Visualizing Digital Discourse: Interactional, Institutional and Ideological Perspectives","publisher":"De Gruyter Mouton","issn":"","isbn":"978-1-5015-1011-3","conference_name":"","conference_place":"","conference_date":"","last_updated_cnr":"2025-01-29 23:22:07","last_updated_oai":"2025-01-29 23:22:07","last_updated_www":"0000-00-00 00:00:00"},{"id":355,"id_source":422985,"institutes":["ILC"],"type":"conference_misc","type_order":8,"title":"An internationally FAIR Mediated Digital Discourse Corpus: towards scientific and pedagogical reuse","year":2020,"authors":["Panckhurst, R.","Frontini, F."],"authors_source":"Panckhurst, Rachel; Frontini, Francesca","authors_cnr_name":["FRONTINI, FRANCESCA"],"authors_cnr_id":["rp02790"],"authors_cnr_institute":[],"abstract":"In this paper, the authors present a French Mediated Digital Discourse corpus, (88milSMS http: \/\/88milsms. huma-num. fr https: \/\/hdl. handle. net\/11403\/comere\/ cmr-88milsms). Efforts were undertaken over the years to ensure its publication according to the best practices and standards of the community, thus guaranteeing compliance with FAIR principles and CLARIN recommendations with pertinent scientific and pedagogical reuse","keywords":["FAIR data","SMS corpus"],"pages":"","url":"https:\/\/www.clarin.eu\/clarin-annual-conference-2020-abstracts","volume":"","doi":"","editors":["Navarretta, C.","Eskevich, M."],"editors_source":"Costanza Navarretta, Maria Eskevich","published":"Proceedings of CLARIN Annual Conference 2020 (5-7 October). Virtual Edition","publisher":"","issn":"","isbn":"","conference_name":"CLARIN Annual Conference 2020 (5-7 October). Virtual Edition","conference_place":"","conference_date":"","last_updated_cnr":"2026-03-04 01:27:54","last_updated_oai":"2026-03-04 01:27:54","last_updated_www":"0000-00-00 00:00:00"},{"id":1074,"id_source":384006,"institutes":["ILC"],"type":"misc","type_order":11,"title":"Dans les coulisses des infrastructures europ\u00e9ennes en SHS. R\u00f4le et opportunit\u00e9s pour les acteurs de la recherche (ing\u00e9nieurs et chercheurs)","year":2020,"authors":["Frontini, F."],"authors_source":"Frontini; Francesca","authors_cnr_name":["FRONTINI, FRANCESCA","FRONTINI, FRANCESCA"],"authors_cnr_id":["rp02790","rp02790"],"authors_cnr_institute":[],"abstract":"La composante technologique prend une dimension de jour en jour plus importante en LLASHS. Les projets de recherche sont de plus en plus nombreux \u00e0 mobiliser de gros volumes de donn\u00e9es exigeant des services adapt\u00e9s garants de formes de m\u00e9thodologies augment\u00e9es (exploitation, interop\u00e9rabilit\u00e9, accessibilit\u00e9, archivage). Afin de partager les savoirs et de garantir l'interop\u00e9rabilit\u00e9 et la pr\u00e9servation \u00e0 long terme de ces ressources et services, de grandes infrastructures informatiques se mettent en place aux niveaux national et international. Dans cette pr\u00e9sentation, vous allez d\u00e9couvrir le panorama, en la mati\u00e8re, des e-infrastructures et des grands projets europ\u00e9ens \u00e0 caract\u00e8re infrastructurel, avec un accent particulier sur les technologies utilis\u00e9es, les principaux services offerts, et les aspects les plus int\u00e9ressants en termes de synergie entre approches et disciplines diff\u00e9rentes. La pr\u00e9sentation portera sur des ERICs (European Research Infrastructure Consortium) \u00e9tablis, comme CLARIN et DARIAH, et sur des projets r\u00e9cents ou en cours de d\u00e9veloppement, comme PARTHENOS, SSHOC, ELEXIS et TRIPLE. Concernant les aspects techniques, on abordera les questions li\u00e9es au d\u00e9p\u00f4t, au stockage, \u00e0 l'identification (sigle sign on), aux formats et choix des m\u00e9tadonn\u00e9es et de mod\u00e9lisation formelle, \u00e0 la recherche f\u00e9d\u00e9r\u00e9e des sources. Nous soulignerons en particulier l'interaction de ces projets avec les infrastructures nationales, notamment Huma-Num, ainsi qu'avec la r\u00e9cemment constitu\u00e9e European Open Science Cloud (EOSC). La pr\u00e9sentation aura une vis\u00e9e pratique, avec l'objectif de fournir des indications concr\u00e8tes aux acteurs de la recherche (chercheurs, ing\u00e9nieurs.) qui souhaitent participer \u00e0 ces initiatives et aux groupes de travail qui les animent, ou plus largement favoriser l'acc\u00e8s des chercheurs fran\u00e7ais aux nombreux services et opportunit\u00e9s offerts","keywords":["Infrastrutture di ricerca","Scienze umane e sociali"],"pages":"","url":"https:\/\/ja-mate2020.sciencesconf.org\/data\/pages\/Resume_Frontini_Nov.pdf","volume":"","doi":"","editors":[],"editors_source":"","published":"","publisher":"","issn":"","isbn":"","conference_name":"Journe\u0301es annuelles du re\u0301seau Mate-shs (JA2020)","conference_place":"","conference_date":"","last_updated_cnr":"2024-03-28 19:56:13","last_updated_oai":"2024-03-28 19:56:13","last_updated_www":"0000-00-00 00:00:00"},{"id":2193,"id_source":389213,"institutes":["ILC"],"type":"conference_misc","type_order":8,"title":"N\u00e9nufar: Modelling a Diachronic Collection of Dictionary Editions as a Computational Lexical Resource","year":2019,"authors":["Bohbot, H.","Frontini, F.","Khan, F.","Khemakhem, M.","Romary, L."],"authors_source":"Herv\u00e9 Bohbot; Francesca Frontini; Fahad Khan; Mohamed Khemakhem; Laurent Romary","authors_cnr_name":["FRONTINI, FRANCESCA"],"authors_cnr_id":["rp02790"],"authors_cnr_institute":[],"abstract":"The Petit Larousse Illustr\u00e9 (PLI) is a monolingual French dictionary which has been published every year since the 1906 edition, and which is therefore a fundamental record of the evolution of the French language. As a consequence of the pre-1948 editions of the PLI entering the public domain in 2018 the N\u00e9nufar (Nouvelle \u00e9dition num\u00e9rique de fac-simil\u00e9s de r\u00e9f\u00e9rence) project was launched at the Praxiling laboratory in Montpellier with the aim of digitizing and making these editions available electronically. The project is still ongoing; various selected editions from each decade are going to be fully digitized (so far the 1906, 1924 and 1925 editions have been completed), and changes backtracked and dated to the specific year. N\u00e9nufar's primary aim is to make the editions available and searchable via an advanced search interface which will not only enable the selective querying of text by lemma and type of content (definitions, examples,.), but crucially also detect and study changes by comparing different editions. In order to do so, a specific web interface has been put in place. Alongside the digitized text, the N\u00e9nufar website contains high quality scans for each page. In compliance with current open data best practices (Wilkinson et al., 2016), the project also aims to make the source data available separately from the querying interface both for research and for A similar project which presents data and scans from subsequent editions of the same legacy dictionary has been carried out by the team behind the Swedish Academy's Wordlist (see Holmer, Malmgren, and Martens (2016) and http: \/\/spraakdata. gu. se\/saolhist\/). eLex 2019: Book of Abstracts 36 long-term preservation. The primary encoding format is TEI-XML; however in our case the TEI encoding is closely inspired by the latest version of the TEI-Lex0 (Ba?ski et al., 2017, Romary & Tasovac, 2018) guidelines for encoding lexicographic resources, which are based upon TEI. The choice of a TEI based approach allows the N\u00e9nufar project to align itself to other pre-existing initiatives and tools. By aligning ourselves to TEI-Lex0 we will be able to make use of digitisation tools such as Grobid (Khemakhem et al., 2017) which have TEI-Lex0 as their native format and which have already been tested and used within the N\u00e9nufar project to speed up the digitization of new editions. In addition we will be able to make use of ongoing initiatives to convert TEI-Lex0 datasets to RDF using the W3C recommendation for publishing lexicons as Linked Data, namely OntoLex-Lemon (McCrae et al., 2017; Bosque-Gil et al., 2016) which will allow for the publication of the N\u00e9nufar dataset as an LOD graph. The LOD version of the N\u00e9nufar dataset, now currently being developed, will be queryable from the available SPARQL endpoint and contain all available editions as one single graph, allowing for expert users to perform complex queries that could detect systematic changes in the dataset. The LOD version is particularly adapted to be linked to other datasets; more recent editions, once added, could also be of interest for NLP applications","keywords":[],"pages":"","url":"https:\/\/iris.cnr.it\/handle\/20.500.14243\/389213","volume":"","doi":"","editors":[],"editors_source":"","published":"","publisher":"","issn":"","isbn":"","conference_name":"","conference_place":"","conference_date":"","last_updated_cnr":"2024-11-29 15:07:33","last_updated_oai":"0000-00-00 00:00:00","last_updated_www":"0000-00-00 00:00:00"},{"id":567,"id_source":376218,"institutes":["ILC"],"type":"conference_article","type_order":7,"title":"One Language to rule them all: modelling Morphological Patterns in a Large Scale Italian Lexicon with SWRL","year":2018,"authors":["Khan, F.","Bellandi, A.","Frontini, F.","Monachini, M."],"authors_source":"Khan, Fahad; Bellandi, Andrea; Frontini, Francesca; Monachini, Monica","authors_cnr_name":["BELLANDI, ANDREA","FRONTINI, FRANCESCA","MONACHINI, MONICA"],"authors_cnr_id":["rp05868","rp02790","rp19457"],"authors_cnr_institute":[],"abstract":"We present an application of Semantic Web Technologies to computational lexicography. More precisely we describe the publication of the morphological layer of the Italian Parole Simple Clips lexicon (PSC-M) as linked open data. The novelty of our work is in the use of the Semantic Web Rule Language (SWRL) to encode morphological patterns, thereby allowing the automatic derivation of the inflectional variants of the entries in the lexicon. By doing so we make these patterns available in a form that is human readable and that therefore gives a comprehensive morphological description of a large number of Italian word","keywords":["Morphology","Linked Open Data","Italian Lexicon","SWRL","SQVRL"],"pages":"4385-4389","url":"http:\/\/www.lrec-conf.org\/proceedings\/lrec2018\/pdf\/844.pdf","volume":"","doi":"","editors":["Chair, N. C. C.","Choukri, K.","Cieri, C.","Declerck, T.","Goggi, S.","Hasida, K.","Isahara, H.","Maegaard, B.","Mariani, J.","Mazo, H.","Moreno, A.","Odijk, J.","Piperidis, S.","Tokunaga, T."],"editors_source":"Nicoletta Calzolari (Conference chair, Khalid Choukri, Christopher Cieri, Thierry Declerck, Sara Goggi, Koiti Hasida, Hitoshi Isahara, Bente Maegaard, Joseph Mariani, He\u0301le\u0300ne Mazo, Asuncion Moreno, Jan Odijk, Stelios Piperidis, Takenobu Tokunaga","published":"Proceedings of the Eleventh International Conference on Language Resources and Evaluation (LREC 2018)","publisher":"European Language Resources Association ELRA (Paris, FRA)","issn":"","isbn":"979-10-95546-00-9","conference_name":"Eleventh International Conference on Language Resources and Evaluation (LREC 2018)","conference_place":"Paris","conference_date":"","last_updated_cnr":"2024-10-08 23:22:39","last_updated_oai":"2024-10-08 23:22:39","last_updated_www":"0000-00-00 00:00:00"},{"id":2229,"id_source":345621,"institutes":["ILC"],"type":"conference_misc","type_order":8,"title":"SWRL your lexicon: adding inflectional rules to a LOD dataset","year":2018,"authors":["Bellandi, A.","Frontini, F.","Khan, F.","Monachini, M."],"authors_source":"Bellandi, Andrea; Frontini, Francesca; Khan, Fahad; Monachini, Monica","authors_cnr_name":["BELLANDI, ANDREA","FRONTINI, FRANCESCA","MONACHINI, MONICA"],"authors_cnr_id":["rp05868","rp02790","rp19457"],"authors_cnr_institute":[],"abstract":"Over the past few years the publication of lexical resources as Linked Data (LD) has taken on ever greater significance within the field of computational lexicography. So far the efforts of the community have been largely directed towards the definition of standards1 and the conversion of single resources (see McCrae et al 2012, Khan et al 2016), but with less of a focus on the technical possibilities afforded by this new mode of publishing lexical data. However, the fact is that the Semantic Web gives us access to a whole ecosystem of standards, languages, and technologies. In this paper we will look at one of these languages, the Semantic Web Rule Language2 (SWRL) and explore whether it might potentially play a useful role in the publication of lexical resources","keywords":[],"pages":"","url":"https:\/\/iris.cnr.it\/handle\/20.500.14243\/345621","volume":"","doi":"","editors":[],"editors_source":"","published":"","publisher":"","issn":"","isbn":"","conference_name":"","conference_place":"","conference_date":"","last_updated_cnr":"2025-02-26 20:40:18","last_updated_oai":"0000-00-00 00:00:00","last_updated_www":"0000-00-00 00:00:00"},{"id":2217,"id_source":345619,"institutes":["ILC"],"type":"conference_misc","type_order":8,"title":"Using Formal Ontologies for the Annotation and Study of Literary Texts","year":2018,"authors":["Khan, A. F. A.","Mugelli, G.","Boschetti, F.","Frontini, F.","Bellandi, A."],"authors_source":"Khan, ANAS FAHAD ASLAM; Mugelli, Gloria; Boschetti, Federico; Frontini, Francesca; Bellandi, Andrea","authors_cnr_name":["KHAN, ANAS FAHAD ASLAM","BOSCHETTI, FEDERICO","FRONTINI, FRANCESCA","BELLANDI, ANDREA"],"authors_cnr_id":["rp05508","rp04876","rp02790","rp05868"],"authors_cnr_institute":[],"abstract":"","keywords":[],"pages":"","url":"https:\/\/iris.cnr.it\/handle\/20.500.14243\/345619","volume":"","doi":"","editors":[],"editors_source":"","published":"","publisher":"","issn":"","isbn":"","conference_name":"","conference_place":"","conference_date":"","last_updated_cnr":"2025-06-14 00:15:52","last_updated_oai":"0000-00-00 00:00:00","last_updated_www":"0000-00-00 00:00:00"},{"id":2228,"id_source":350511,"institutes":["ILC"],"type":"misc","type_order":11,"title":"Parole-Simple-Clip\/Morphological Layer in RDF","year":2018,"authors":["Bellandi, A.","Frontini, F.","Khan, F.","Monachini, M."],"authors_source":"Bellandi, Andrea; Frontini, Francesca; Khan, Fahad; Monachini, Monica","authors_cnr_name":["BELLANDI, ANDREA","FRONTINI, FRANCESCA","MONACHINI, MONICA"],"authors_cnr_id":["rp05868","rp02790","rp19457"],"authors_cnr_institute":[],"abstract":"A version in RDF of the morphological layer of the wide coverage multi-level Italian lexicon Parole-Simple-Clips, containing the parts of speech Noun, Verb, Adjective. The dataset is encoded using the ontolex-lemon vocabulary. Information pertaining to inflectional morphological contained in the original resource is converted into Semantic Web Rule Language (SWRL) rules","keywords":[],"pages":"","url":"https:\/\/iris.cnr.it\/handle\/20.500.14243\/350511","volume":"","doi":"","editors":[],"editors_source":"","published":"","publisher":"","issn":"","isbn":"","conference_name":"","conference_place":"","conference_date":"","last_updated_cnr":"2024-10-08 23:22:18","last_updated_oai":"0000-00-00 00:00:00","last_updated_www":"0000-00-00 00:00:00"},{"id":675,"id_source":339934,"institutes":["ILC"],"type":"book_chapter","type_order":4,"title":"Using SWRL rules to model noun behaviour in Italian","year":2017,"authors":["Khan, F.","Bellandi, A.","Frontini, F.","Monachini, M."],"authors_source":"Khan, F; Bellandi, A; Frontini, F; Monachini, M","authors_cnr_name":["KHAN, ANAS FAHAD ASLAM","BELLANDI, ANDREA","FRONTINI, FRANCESCA","MONACHINI, MONICA"],"authors_cnr_id":["rp05508","rp05868","rp02790","rp19457"],"authors_cnr_institute":[],"abstract":"In this article we describe our ongoing attempts to use the Semantic Web Rule Language (SWRL) to model the morphological layer of a wide-coverage Italian lexical resource, Parole-Simple-Clips (PSC); in this case that subset of PSC dealing with Italian noun morphology. After giving a brief introduction to SWRL and to Italian noun morphology we go onto describe the actual transformation itself. Finally we describe an experiment on our dataset using SWRL rules and queries written in the Semantic Query-Enhanced Rule Web Language (SQWRL)","keywords":["Linked Open Data","Logic Programming","Italian Morphology"],"pages":"134-142","url":"http:\/\/www.scopus.com\/record\/display.url?eid=2-s2.0-85021186095&origin=inward","volume":"","doi":"10.1007\/978-3-319-59888-8_11","editors":["Gracia, J.","Bond, F.","McCrae, J.","Buitelaar, P.","Chiarcos, C.","Hellmann, S."],"editors_source":"Gracia, J; Bond, F; McCrae, JP; Buitelaar, P; Chiarcos, C; Hellmann, S","published":"LANGUAGE, DATA, AND KNOWLEDGE, LDK","publisher":"Springer (Berlin, DEU)","issn":"","isbn":"","conference_name":"","conference_place":"Berlin","conference_date":"","last_updated_cnr":"2025-02-09 22:18:36","last_updated_oai":"2025-02-09 22:18:36","last_updated_www":"0000-00-00 00:00:00"},{"id":491,"id_source":333614,"institutes":["ILC"],"type":"book_chapter","type_order":4,"title":"Semantic Search Engine for Data Management and Sustainable Development: Marine Planning Service Platform","year":2017,"authors":["R Manzella, G. M.","Bartolini, R.","Bustaffa, F.","D'Angelo, P.","De Mattei, M.","Frontini, F.","Maltese, M.","Medone, D.","Monachini, M.","Novellino, A.","Spada, A."],"authors_source":"R Manzella, Giuseppe M; Bartolini, Roberto; Bustaffa, Franco; D'Angelo, Paolo; De Mattei, Maurizio; Frontini, Francesca; Maltese, Maurizio; Medone, Daniele; Monachini, Monica; Novellino, Antonio; Spada, Andrea","authors_cnr_name":["BARTOLINI, ROBERTO","FRONTINI, FRANCESCA","MONACHINI, MONICA"],"authors_cnr_id":["rp00239","rp02790","rp19457"],"authors_cnr_institute":[],"abstract":"This chapter presents a computer platform supporting a Marine Information and Knowledge System based on a repository that gathers, classify and structures marine scientific literature and data, guaranteeing their accessibility by means of standard protocols. This requires the access to quality controlled data and to information that is provided in grey literature and\/or in relevant scientific literature. There exist efforts to develop search engines to find author's contributions to scientific literature or publications. This implies the use of persistent identifiers. However very few efforts are dedicated to link publications to data that was used, or cited in them or that can be of importance for the published studies. Full-text technologies are often unsuccessful since they assume the presence of specific keywords in the text; to fix this problem, it is suggested to use different semantic technologies for retrieving the text and data and thus getting much more complying results","keywords":["Marine Information and Knowledge System"],"pages":"127-154","url":"http:\/\/www.igi-global.com\/chapter\/semantic-search-engine-for-data-management-and-sustainable-development\/166839#","volume":"","doi":"10.4018\/978-1-5225-0700-0.ch006","editors":["Paolo Diviacco, A. L.","Glaves, H."],"editors_source":"Paolo Diviacco, Adam Leadbetter & Helen Glaves","published":"Oceanographic and Marine Cross-Domain Data Management for Sustainable Development","publisher":"IGI Global (Hershey, USA)","issn":"5225-0700","isbn":"","conference_name":"","conference_place":"Hershey","conference_date":"","last_updated_cnr":"2024-05-17 17:44:39","last_updated_oai":"2024-05-17 17:44:39","last_updated_www":"0000-00-00 00:00:00"},{"id":2278,"id_source":342012,"institutes":["ILC"],"type":"conference_article","type_order":7,"title":"Situating Word Senses in their Historical Context with Linked Data","year":2017,"authors":["Khan, F.","Bowers, J.","Frontini, F."],"authors_source":"Fahad Khan; Jack Bowers;Francesca Frontini","authors_cnr_name":["FRONTINI, FRANCESCA"],"authors_cnr_id":["rp02790"],"authors_cnr_institute":[],"abstract":"In this article we present a Semantic Web-based model for creating lexical resources in which the diachronic and, more broadly, contextual dimensions of word meaning can be explicitly represented as part of a graph-based data structure. We start by discussing why Linked Data is the right publishing approach for such diachronic datasets. We then describe our model, lemonEty, which utilizes the ontology engineering technique of perdurants in order to model lexical entries as dynamic processes. Next we go onto explain how to represent etymologies using our model, and in particular how to associate temporal information with word senses, taking examples from two different lexicographic resources. In addition, we will show how our model deals with cognates and attestations","keywords":[],"pages":"","url":"https:\/\/iris.cnr.it\/handle\/20.500.14243\/342012","volume":"","doi":"","editors":[],"editors_source":"","published":"","publisher":"","issn":"","isbn":"","conference_name":"","conference_place":"","conference_date":"","last_updated_cnr":"2024-11-29 14:59:20","last_updated_oai":"0000-00-00 00:00:00","last_updated_www":"0000-00-00 00:00:00"},{"id":1492,"id_source":320996,"institutes":["ILC"],"type":"journal_article","type_order":1,"title":"A semantic engine for grey literature retrieval in the oceanography domain","year":2016,"authors":["Goggi, S.","Pardelli, G.","Bartolini, R.","Frontini, F.","Monachini, M.","Manzella, G.","De Mattei, M.","Bustaffa, F."],"authors_source":"Goggi, Sara; Pardelli, Gabriella; Bartolini, Roberto; Frontini, Francesca; Monicamonachini, ; Manzella, Giuseppe; De Mattei, Maurizio; Bustaffa, Franco","authors_cnr_name":["GOGGI, SARA","PARDELLI, GABRIELLA","BARTOLINI, ROBERTO","FRONTINI, FRANCESCA","MONACHINI, MONICA"],"authors_cnr_id":["rp20855","rp20820","rp00239","rp02790","rp19457"],"authors_cnr_institute":[],"abstract":"Here we present the final results of the MAPS (Marine Planning and Service Platform) project, an environment designed for gathering, classifying, managing and accessing marine scientific literature and data, making it available for search to Operative Oceanography researchers of various institutions by means of standard protocols. The system takes as input non-textual data (measurements) and text-both published papers and documentation-and it provides an advanced search facility thanks to the rich set of metadata and, above all, to the possibility of a refined and domain targeted key-word indexing of texts using Natural Language Processing (NLP) techniques. The paper describes the system in its details providing also evidence of evaluation","keywords":["Information Extraction","Search Engine","Operative Oceanography"],"pages":"155-161","url":"http:\/\/www.greynet.org\/thegreyjournal\/currentissue.html","volume":"12 (3)","doi":"","editors":[],"editors_source":"","published":"THE GREY JOURNAL","publisher":"","issn":"1574-1796","isbn":"","conference_name":"","conference_place":"","conference_date":"","last_updated_cnr":"2025-02-07 06:08:39","last_updated_oai":"2025-02-07 06:08:39","last_updated_www":"0000-00-00 00:00:00"},{"id":2305,"id_source":322086,"institutes":["ILC"],"type":"journal_article","type_order":1,"title":"Restructuring a Taxonomy of Literary Themes and Motifs for More Efficient Querying","year":2016,"authors":["Khan, F.","Arrigoni, S.","Boschetti, F.","Frontini, F."],"authors_source":"Fahad Khan; Silvia Arrigoni; Federico Boschetti; Francesca Frontini","authors_cnr_name":["BOSCHETTI, FEDERICO","FRONTINI, FRANCESCA"],"authors_cnr_id":["rp04876","rp02790"],"authors_cnr_institute":[],"abstract":"In this paper we describe ongoing work in the restructuring of a tagset originally organised as a taxonomy and used to annotate literary themes and motifs in a corpus of classical works of poetry from a number of different traditions. We show how such a tagset can be rendered more efficient and useful through the appropriation of ideas and techniques from lexical semantics and ontology design. The newly redesigned tagset is described with examples showing how the new design is much more expressive than the old taxonomy; furthermore, an example query is described in order to demonstrate how more refined semantic searches can be carried using the new version of the taxonomy. The final result is, we hope, a resource that will be useful not only for the specific project for which it was developed but one that is well-designed and well-documented enough to be of use for other similar semantic annotation tasks","keywords":[],"pages":"","url":"https:\/\/iris.cnr.it\/handle\/20.500.14243\/322086","volume":"","doi":"10.14195\/2182-8830","editors":[],"editors_source":"","published":"MATLIT","publisher":"","issn":"2182-8830","isbn":"","conference_name":"","conference_place":"","conference_date":"","last_updated_cnr":"2024-03-11 03:39:03","last_updated_oai":"0000-00-00 00:00:00","last_updated_www":"0000-00-00 00:00:00"},{"id":1158,"id_source":327645,"institutes":["ILC"],"type":"journal_article","type_order":1,"title":"CLARIN, l'infrastruttura europea delle risorse linguistiche per le scienze umane e sociali e il suo network italiano CLARIN-IT","year":2016,"authors":["Monachini, M.","Frontini, F."],"authors_source":"Monachini, Monica; Frontini, Francesca","authors_cnr_name":["MONACHINI, MONICA","FRONTINI, FRANCESCA"],"authors_cnr_id":["rp19457","rp02790"],"authors_cnr_institute":[],"abstract":"ll 1\u00b0ottobre 2015 il MIUR firma l'adesione dell'Italia a CLARIN-ERIC, l'infrastruttura di ricerca che offre risorse e tecnologie linguistiche dedicate al settore delle scienze del linguaggio e delle scienze umane e sociali. Questo articolo intende fornire alla comunit\u00e0 italiana una ampia panoramica di CLARIN, la sua missione, i suoi pilastri, i servizi, la sua organizzazione tecnica ed amministrativa e la struttura di governance, sia a livello europeo che locale. Viene introdotto il network italiano, con il primo centro nazionale ILC4CLARIN, ospitato ed in via di sviluppo presso l'ILC-CNR, le funzionalit\u00e0, le risorse ed i servizi offerti; viene presentato infine il primo nucleo del consorzio nazionale CLARIN-IT, illustrando i criteri di costituzione, le attivit\u00e0 previste e le prospettive future","keywords":["Infrastrutture di ricerca","Tecnologie linguistiche","Network italiano CLARIN-IT"],"pages":"1-30","url":"http:\/\/www.ai-lc.it\/IJCoL\/v2n2\/1-monachini_and_frontini.pdf","volume":"VOL. 2 (2)","doi":"","editors":[],"editors_source":"","published":"IJCOL","publisher":"","issn":"2499-4553","isbn":"","conference_name":"","conference_place":"","conference_date":"","last_updated_cnr":"2024-11-28 00:34:30","last_updated_oai":"2024-11-28 00:34:30","last_updated_www":"0000-00-00 00:00:00"},{"id":1663,"id_source":320992,"institutes":["ILC"],"type":"book_chapter","type_order":4,"title":"GeoDomainWordNet: Linking the Geonames Ontology to WordNet","year":2016,"authors":["Frontini, F.","Del Gratta, R.","Monachini, M."],"authors_source":"Frontini, Francesca; DEL GRATTA, Riccardo; Monachini, Monica","authors_cnr_name":["FRONTINI, FRANCESCA","DEL GRATTA, RICCARDO","MONACHINI, MONICA"],"authors_cnr_id":["rp02790","rp00284","rp19457"],"authors_cnr_institute":[],"abstract":"This paper illustrates the transformation of GeoNames' ontology concepts, with their English labels and glosses, into a GeoDomain WordNet-like resource in English, its translation into Italian, and its linking to the existing generic WordNets of both languages. The paper describes the criteria used for the linking of domain synsets to each other and to the generic ones and presents the published resource in RDF according to the w3c and lemon schema","keywords":["GeoNames","WordNet","Language resources","Lexi","Linguistic linked data","lemon","RDF"],"pages":"229-242","url":"http:\/\/link.springer.com\/chapter\/10.1007\/978-3-319-43808-5_18","volume":"","doi":"10.1007\/978-3-319-43808-5","editors":[],"editors_source":"","published":"Human Language Technology. Challenges for Computer Science and Linguistics","publisher":"","issn":"","isbn":"978-3-319-43808-5","conference_name":"","conference_place":"","conference_date":"","last_updated_cnr":"2025-06-15 00:33:48","last_updated_oai":"2025-06-15 00:33:48","last_updated_www":"0000-00-00 00:00:00"},{"id":1870,"id_source":324185,"institutes":["ILC"],"type":"edited_volume","type_order":5,"title":"Language and Ontology (LangOnto2) & Terminology and Knowledge Structures (TermiKS)","year":2016,"authors":["Khan, F.","Vintar, P.","Ara\u00faz, P. L.","Faber, P.","Frontini, F.","Parvizi, A.","Grisimeunovi, L.","Unger, C."],"authors_source":"Fahad Khan; pela Vintar ; Pilar Le\u00f3n Ara\u00faz; Pamela Faber; Francesca Frontini; Artemis Parvizi; Larisa GriSimeunovi; Christina Unger","authors_cnr_name":["FRONTINI, FRANCESCA"],"authors_cnr_id":["rp02790"],"authors_cnr_institute":[],"abstract":"This joint workshop brings together two different but closely related strands of research. On the one hand it looks at the overlap between ontologies and computational linguistics and on the other it explores the relationship between knowledge modelling and terminologies. In particular the workshop aims to create a forum for discussion in which the different relationships and commonalities between these two areas can be explored in detail, as well as presenting cutting edge research in each of the two individual areas. A significant amount of human knowledge can be found in texts. It is not surprising that languages such as OWL, which allow us to formally represent this knowledge, have become more and more popular both in linguistics and in automated language processing. For instance ontologies are now of core interest to many NLP fields including Machine Translation, Question Answering, Text Summarization, Information Retrieval, and Word Sense Disambiguation. At a more abstract level, however, ontologies can also help us to model and reason about phenomena in natural language semantics. In addition, ontologies and taxonomies can also be used in the organisation and formalisation of linguistically relevant categories such as those used in tagsets for corpus annotation. Notably also, the fact that formal ontologies are being increasingly accessed by users with limited to no background in formal logic has led to a growing interest in developing accessible front ends that allow for easy querying and summarisation of ontologies. It has also led to work in developing natural language interfaces for authoring ontologies and evaluating their design. Additionally in recent years there has been a renewed interest in the linguistic aspects of accessing, extracting, representing, modelling and transferring knowledge. Numerous tools for the automatic extraction of terms, term variants, knowledge-rich contexts, definitions, semantic relations and taxonomies from specialized corpora have been developed for a number of languages, and new theoretical approaches have emerged as potential frameworks for the study of specialized communication. However, the building of adequate knowledge models for practitioners (e. g. experts, researchers, translators, teachers etc.), on the one hand, and NLP applications (including cross-language, cross-domain, cross-device, multi-modal, multi-platform applications), on the other hand, still remains a challenge. The papers included in the workshop range across a wide variety of different areas and reflect the strong inter-disciplinary approach, which characterises both areas of research. In addition we are very happy to include two invited talks in the program presented by authorities in their respective fields: Pamela Faber from the field of terminology, and John McCrae, an expert on linguistic linked data and the interface between NLP and ontologies","keywords":["lexicons","ontologies"],"pages":"","url":"http:\/\/www.lrec-conf.org\/proceedings\/lrec2016\/index.html","volume":"","doi":"","editors":[],"editors_source":"","published":"","publisher":"","issn":"","isbn":"","conference_name":"","conference_place":"","conference_date":"","last_updated_cnr":"2024-04-28 10:43:26","last_updated_oai":"2024-04-28 10:43:26","last_updated_www":"0000-00-00 00:00:00"},{"id":1686,"id_source":324176,"institutes":["ILC"],"type":"conference_article","type_order":7,"title":"LREC as a Graph: People and Resources in a Network","year":2016,"authors":["Del Gratta, R.","Frontini, F.","Monachini, M.","Pardelli, G.","Russo, I.","Bartolini, R.","Khan, F.","Soria, C.","Calzolari, N."],"authors_source":"Del Gratta, R.; Frontini, F.; Monachini, M.; Pardelli, G.; Russo, I.; Bartolini, R.; Khan, F.; Soria, C.; Calzolari, N.","authors_cnr_name":["DEL GRATTA, RICCARDO","FRONTINI, FRANCESCA","MONACHINI, MONICA","PARDELLI, GABRIELLA","RUSSO, IRENE","BARTOLINI, ROBERTO","KHAN, ANAS FAHAD ASLAM","SORIA, CLAUDIA"],"authors_cnr_id":["rp00284","rp02790","rp19457","rp20820","rp02389","rp00239","rp05508","rp17652"],"authors_cnr_institute":[],"abstract":"This proposal describes a new way to visualise resources in the LREMap, a community-built repository of language resource descriptions and uses. The LREMap is represented as a force-directed graph, where resources, papers and authors are nodes. The analysis of the visual representation of the underlying graph is used to study how the community gathers around LRs and how LRs are used in research","keywords":["Language Resources","Resources Documentation","Data Visualisation"],"pages":"2529-2532","url":"http:\/\/www.lrec-conf.org\/proceedings\/lrec2016\/index.html","volume":"","doi":"","editors":["Calzolari, N.","Choukri, K.","Declerck, T.","Goggi, S.","Grobelnik, M.","Maegaard, B.","Mariani, J.","Mazo, H.","Moreno, A.","Odijk, J.","Piperidis, S."],"editors_source":"Nicoletta Calzolari (Conference Chair), Khalid Choukri, Thierry Declerck, Sara Goggi, Marko Grobelnik, Bente Maegaard, Joseph Mariani, He\u0301le\u0300ne Mazo, Asuncio\u0301n Moreno, Jan Odijk, Stelios Piperidis","published":"Tenth International Conference on Language Resources and Evaluation (LREC 2016)","publisher":"European Language Resources Association ELRA (Paris, FRA)","issn":"","isbn":"978-2-9517408-9-1","conference_name":"Tenth International Conference on Language Resources and Evaluation (LREC 2016)","conference_place":"Paris","conference_date":"","last_updated_cnr":"2024-12-04 00:24:54","last_updated_oai":"2024-12-04 00:24:54","last_updated_www":"0000-00-00 00:00:00"},{"id":601,"id_source":315259,"institutes":["ILC"],"type":"conference_article","type_order":7,"title":"A semantic engine for grey literature retrieval in the oceanography domain","year":2016,"authors":["Goggi, S.","Pardelli, G.","Bartolini, R.","Frontini, F.","Monachini, M.","Manzella, G.","De Mattei, M.","Bustaffa, F."],"authors_source":"Sara Goggi; Gabriella Pardelli; Roberto Bartolini; Francesca Frontini; MonicaMonachini; Giuseppe Manzella; Maurizio De Mattei;Franco Bustaffa","authors_cnr_name":["GOGGI, SARA","PARDELLI, GABRIELLA","BARTOLINI, ROBERTO","FRONTINI, FRANCESCA","MONACHINI, MONICA"],"authors_cnr_id":["rp20855","rp20820","rp00239","rp02790","rp19457"],"authors_cnr_institute":[],"abstract":"Here we present the final results of the MAPS (Marine Planning and Service Platform) project, an environment designed for gathering, classifying, managing and accessing marine scientific literature and data, making it available for search to Operative Oceanography researchers of various institutions by means of standard protocols. The system takes as input non-textual data (measurements) and text-both published papers and documentation-and it provides an advanced search facility thanks to the rich set of metadata and, above all, to the possibility of a refined and domain targeted key-word indexing of texts using Natural Language Processing (NLP) techniques. The paper describes the system in its details providing also evidence of evaluation","keywords":["Information Extraction","Search Engine","Operative Oceanography"],"pages":"104-111","url":"https:\/\/iris.cnr.it\/handle\/20.500.14243\/315259","volume":"","doi":"","editors":["Farace, D.","Frantzen, J."],"editors_source":"Dominic Farace, Jerry Frantzen","published":"","publisher":"","issn":"","isbn":"978-90-77484-27-2","conference_name":"Seventeenth International Conference on Grey Literature. A New Wave of Textual and Non-Textual Grey Literature","conference_place":"","conference_date":"","last_updated_cnr":"2025-02-07 03:29:08","last_updated_oai":"2025-02-07 03:29:08","last_updated_www":"0000-00-00 00:00:00"},{"id":2307,"id_source":322088,"institutes":["ILC"],"type":"conference_article","type_order":7,"title":"Leveraging a narrative ontology to query a literary text","year":2016,"authors":["Khan, A. F. A.","Bellandi, A.","Benotto, G.","Frontini, F.","Giovannetti, E.","Reboul, M."],"authors_source":"Khan, ANAS FAHAD ASLAM; Khan, ANAS FAHAD ASLAM; Bellandi, Andrea; Benotto, Giulia; Frontini, Francesca; Giovannetti, Emiliano; Reboul, Marianne","authors_cnr_name":["KHAN, ANAS FAHAD ASLAM","KHAN, ANAS FAHAD ASLAM","BELLANDI, ANDREA","BENOTTO, GIULIA","FRONTINI, FRANCESCA","GIOVANNETTI, EMILIANO"],"authors_cnr_id":["rp05508","rp05508","rp05868","rp06562","rp02790","rp21296"],"authors_cnr_institute":[],"abstract":"In this work we propose a model for the representation of the narrative of a literary text. The model is structured in an ontology and a lexicon constituting a knowledge base that can be queried by a system. This narrative ontology, as well as describing the actors, locations, situations found in the text, provides an explicit formal representation of the timeline of the story. We will focus on a specific case study, that of the representation of a selected portion of Homer's Odyssey, in particular of the knowledge required to answer a selection of salient queries, formulated by a literary scholar. This work is being carried out within the framework of the Semantic Web by adopting models and standards such as RDF, OWL, SPARQL, and lemon among others","keywords":[],"pages":"","url":"https:\/\/iris.cnr.it\/handle\/20.500.14243\/322088","volume":"","doi":"10.4230\/OASIcs.CMN.2016.10","editors":[],"editors_source":"","published":"","publisher":"","issn":"","isbn":"9783959770200","conference_name":"","conference_place":"","conference_date":"","last_updated_cnr":"2025-06-14 00:42:34","last_updated_oai":"0000-00-00 00:00:00","last_updated_www":"0000-00-00 00:00:00"},{"id":2315,"id_source":322106,"institutes":["ILC"],"type":"conference_article","type_order":7,"title":"Converting the Liddell Scott Greek-English Lexicon into Linked Open Data using lemon","year":2016,"authors":["Khan, F.","Frontini, F.","Boschetti, F.","Monachini",", M."],"authors_source":"Khan F; Frontini F; Boschetti F; Monachini; M","authors_cnr_name":["KHAN, ANAS FAHAD ASLAM","FRONTINI, FRANCESCA","BOSCHETTI, FEDERICO","MONACHINI, MONICA"],"authors_cnr_id":["rp05508","rp02790","rp04876","rp19457"],"authors_cnr_institute":[],"abstract":"The emergence and growing popularity of Linked Open Data (LOD) offers researchers a new range of possibilities when it comes to publishing datasets online (Hyv\u00f6nen 2012, Oomen et al 2012); indeed not only does the success of LOD greatly facilitate the process of making scholarly data accessible and to a wider community but it also permits the enrichment of individual datasets by linking them to the other datasets available on the so called Linked Open Data Cloud. The advantages of Linked Open Data for teachers, academics and students in the humanities are obvious and are indeed manifold. However there is currently a paucity of linked open datasets in fields such as philology and literary studies, and in particular of datasets that deal with classical languages such as ancient Greek, Sanskrit, and Latin. This seems strange given the rich abundance of surviving works, of both a religious and secular character, that exist in those languages. A salient consideration here relates to the fact that even when such works have been digitised and made available in a format such as TEI-XML, a format which renders the structure and content of such texts more amenable to computer processing, the conversion of these resources into the Resource Data Framework (RDF), the standardised data model that underpins the Semantic Web, is not always straightforward. In this article we describe ongoing work in the conversion of an important 19th century Ancient Greek resource the Liddell-Scott-Jones Lexicon, into RDF, part of a wider program of work that has been recently initiated at CNR-ILC in converting historical lexicons in languages such as Greek, Latin and Arabic into Linked Open Data","keywords":[],"pages":"","url":"https:\/\/iris.cnr.it\/handle\/20.500.14243\/322106","volume":"","doi":"","editors":[],"editors_source":"","published":"","publisher":"","issn":"","isbn":"978-83-942760-3-4","conference_name":"","conference_place":"","conference_date":"","last_updated_cnr":"2025-01-21 22:39:59","last_updated_oai":"0000-00-00 00:00:00","last_updated_www":"0000-00-00 00:00:00"},{"id":489,"id_source":324187,"institutes":["ILC"],"type":"conference_article","type_order":7,"title":"Al Qamus al Muhit, a Medieval Arabic Lexicon in LMF","year":2016,"authors":["Nahli, O.","Frontini, F.","Monachini, M.","Khan, F.","Zarghili, A.","Khalfi, M."],"authors_source":"Nahli, O; Frontini, F; Monachini, M; Khan, F; Zarghili, A; Khalfi, M","authors_cnr_name":["NAHLI, OUAFAE","FRONTINI, FRANCESCA","MONACHINI, MONICA","KHAN, ANAS FAHAD ASLAM"],"authors_cnr_id":["rp03551","rp02790","rp19457","rp05508"],"authors_cnr_institute":[],"abstract":"This paper describes the conversion into LMF, a standard lexicographic digital format of 'al-q?m?s al-mu???, a Medieval Arabic lexicon. The lexicon is first described, then all the steps required for the conversion are illustrated. The work is will produce a useful lexicographic resource for Arabic NLP, but is also interesting per se, to study the implications of adapting the LMF model to the Arabic language. Some reflections are offered as to the status of roots with respect to previously suggested representations. In particular, roots are, in our opinion are to be not treated as lexical entries, but modeled as lexical metadata for classifying and identifying lexical entries. In this manner, each root connects all entries that are derived from it","keywords":["Arabic Lexicon","LMF","Al Qamus al Muhi"],"pages":"943-950","url":"http:\/\/www.lrec-conf.org\/proceedings\/lrec2016\/index.html","volume":"","doi":"","editors":["Calzolari, N.","Choukri, K.","Declerck, T.","Goggi, S.","Grobelnik, M.","Maegaard, B.","Mariani, J.","Mazo, H.","Moreno, A.","Odijk, J.","Piperidis, S."],"editors_source":"Nicoletta Calzolari (Conference Chair), Khalid Choukri, Thierry Declerck, Sara Goggi, Marko Grobelnik, Bente Maegaard, Joseph Mariani, He\u0301le\u0300ne Mazo, Asuncio\u0301n Moreno, Jan Odijk, Stelios Piperidis","published":"","publisher":"European Language Resources Association ELRA (Paris, FRA)","issn":"","isbn":"978-2-9517408-9-1","conference_name":"Tenth International Conference on Language Resources and Evaluation (LREC 2016)","conference_place":"Paris","conference_date":"","last_updated_cnr":"2025-01-24 23:19:03","last_updated_oai":"2025-01-24 23:19:03","last_updated_www":"0000-00-00 00:00:00"},{"id":883,"id_source":324227,"institutes":["ILC"],"type":"conference_misc","type_order":8,"title":"Marine Planning and Service Platform: Specific Ontology Based semantic Search Engine Serving Data Management and Sustainable Development","year":2016,"authors":["Manzella, G.","Bartolini, R.","Bustaffa, F.","D'Angelo, P.","De Mattei, M.","Frontini, F.","Maltese, M.","Medone, D.","Monachini, M.","Novellino, A.","Spada, A."],"authors_source":"Manzella Giuseppe, Mr; Bartolini, Roberto; Bustaffa, Franco; D'Angelo, Paolo; De Mattei, Maurizio; Frontini, Francesca; Maltese, Maurizio; Medone, Daniele; Monachini, Monica; Novellino, Antonio; Spada, Andrea","authors_cnr_name":["BARTOLINI, ROBERTO","FRONTINI, FRANCESCA","MONACHINI, MONICA"],"authors_cnr_id":["rp00239","rp02790","rp19457"],"authors_cnr_institute":[],"abstract":"The MAPS (Marine Planning and Service Platform) project is aiming at building a computer platform supporting a Marine Information and Knowledge System. One of the main objective of the project is to develop a repository that should gather, classify and structure marine scientific literature and data thus guaranteeing their accessibility to researchers and institutions by means of standard protocols. In oceanography the cost related to data collection is very high and the new paradigm is based on the concept to collect once and re-use many times (for re-analysis, marine environment assessment, studies on trends, etc). This concept requires the access to quality controlled data and to information that is provided in reports (grey literature) and\/or in relevant scientific literature. Hence, creation of new technology is needed by integrating several disciplines such as data management, information systems, knowledge management","keywords":["Marine Information","Knowledge System"],"pages":"2","url":"http:\/\/meetingorganizer.copernicus.org\/EGU2016\/orals\/20144","volume":"18","doi":"","editors":[],"editors_source":"","published":"","publisher":"","issn":"1607-7962","isbn":"","conference_name":"European Geosciences Union General Assembly (EGU 2016)","conference_place":"","conference_date":"","last_updated_cnr":"2024-06-08 11:56:25","last_updated_oai":"2024-06-08 11:56:25","last_updated_www":"0000-00-00 00:00:00"},{"id":2069,"id_source":333124,"institutes":["ILC"],"type":"misc","type_order":11,"title":"CLARIN-IT: servizi per la comunit\u00e0 italiana delle scienze umane e sociali","year":2016,"authors":["Monachini, M.","Enea, A.","Frontini, F."],"authors_source":"Monachini Monica; Alessandro Enea; Francesca Frontini","authors_cnr_name":["MONACHINI, MONICA","ENEA, ALESSANDRO","FRONTINI, FRANCESCA"],"authors_cnr_id":["rp19457","rp19059","rp02790"],"authors_cnr_institute":[],"abstract":"CLARIN-IT-The Italian Common Language Resources and Technology Infrastructure: Monica Monachini-CLARIN Italian National Coordinator Alessandro Enea-Responsible of ILCforCLARIN & contact person for IDEM Francesca Frontini-Standing Committee for CLARIN Technical Centres (SCCTC) ILC-CNR National Representative","keywords":["CLARIN-IT","The Italian Common Language Resources and Technology Infrastructure"],"pages":"","url":"http:\/\/www.clarin-it.it\/en\/content\/clarin-it-idem-day-2016","volume":"","doi":"","editors":[],"editors_source":"","published":"","publisher":"","issn":"","isbn":"","conference_name":"CLARIN-IT @ IDEM Day 2016","conference_place":"","conference_date":"","last_updated_cnr":"2024-06-07 14:51:57","last_updated_oai":"2024-06-07 14:51:57","last_updated_www":"0000-00-00 00:00:00"},{"id":725,"id_source":222847,"institutes":["ILC"],"type":"journal_article","type_order":1,"title":"Converting the PAROLE SIMPLE CLIPS Lexicon into RDF with lemon","year":2015,"authors":["Del Gratta, R.","Frontini, F.","Khan, F.","Monachini, M."],"authors_source":"Del Gratta Riccardo; Francesca Frontini; Fahad Khan; Monica Monachini","authors_cnr_name":["DEL GRATTA, RICCARDO","FRONTINI, FRANCESCA","MONACHINI, MONICA"],"authors_cnr_id":["rp00284","rp02790","rp19457"],"authors_cnr_institute":[],"abstract":"This paper describes the publication and linking of (parts of) PAROLE SIMPLE CLIPS (PSC), a large scale Italian lexicon, to the Semantic Web and the Linked Data cloud using the lemon model. The main challenge of the conversion is discussed, namely the reconciliation between the PSC semantic structure which contains richly encoded semantic information, following the qualia structure of the Generative Lexicon theory and the lemon view of lexical sense as a reified pairing of a lexical item and a concept in an ontology. The result is two datasets: one consists of a list of lemon lexical entries with their lexical properties, relations and senses; the other consists of a list of OWL individuals representing the referents for the lexical senses. These OWL individuals are linked to each other by a set of semantic relations and mapped onto the SIMPLE OWL ontology of higher level semantic types","keywords":["lemon","linked data","generative lexicon","RDF","OWL","lexical resource"],"pages":"387-392","url":"http:\/\/www.semantic-web-journal.net\/content\/converting-parole-simple-clips-lexicon-rdf-lemon-0","volume":"6","doi":"10.3233\/SW-140168","editors":[],"editors_source":"","published":"SEMANTIC WEB (PRINT)","publisher":"","issn":"1570-0844","isbn":"","conference_name":"","conference_place":"","conference_date":"","last_updated_cnr":"2025-04-25 00:31:09","last_updated_oai":"2025-04-25 00:31:09","last_updated_www":"0000-00-00 00:00:00"},{"id":652,"id_source":296111,"institutes":["ILC"],"type":"journal_article","type_order":1,"title":"Marine Planning and Service Platform (MAPS) An Advanced Research Engine for Grey Literature in Marine Science","year":2015,"authors":["Goggi, S.","Monachini, M.","Frontini, F.","Bartolini, R.","Pardelli, G.","De Mattei, M.","Bustaffa, F.","Manzella, G."],"authors_source":"Sara Goggi; Monica Monachini; Francesca Frontini; Roberto Bartolini; Gabriella Pardelli; Maurizio De Mattei; Franco Bustaffa;Giuseppe Manzella","authors_cnr_name":["GOGGI, SARA","MONACHINI, MONICA","FRONTINI, FRANCESCA","BARTOLINI, ROBERTO","PARDELLI, GABRIELLA"],"authors_cnr_id":["rp20855","rp19457","rp02790","rp00239","rp20820"],"authors_cnr_institute":[],"abstract":"The MAPS (Marine Planning and Service Platform) project is a development of the Marine project (Ricerca Industriale e Sviluppo Sperimentale Regione Liguria 2007-2013) aiming at building a computer platform for supporting a Marine Information and Knowledge System, as part of the data management activities. One of the main objective of the project is to develop a repository that should gather, classify and structure marine scientific literature and data thus guaranteeing their accessibility to researchers and institutions by means of standard protocols. We will present the scenario of the Operative Oceanography together with the technologies used to develop an advanced search engine which aims at providing rapid and efficient access to a Digital Library of oceanographic data. The case-study is also highlighting how the retrieval of grey literature from this specific marine community could be reproduced for similar communities as well, thus revealing the great impact that the processing, re-use as well as application of grey data have on societal needs\/problems and their answers","keywords":["Marine Science","Search Engine","Source Data","Oceanography"],"pages":"171-178","url":"https:\/\/iris.cnr.it\/handle\/20.500.14243\/296111","volume":"11 (3)","doi":"","editors":[],"editors_source":"","published":"THE GREY JOURNAL","publisher":"","issn":"1574-1796","isbn":"","conference_name":"","conference_place":"","conference_date":"","last_updated_cnr":"2024-06-19 07:23:08","last_updated_oai":"2024-06-19 07:23:08","last_updated_www":"0000-00-00 00:00:00"},{"id":705,"id_source":292095,"institutes":["ILC"],"type":"book_chapter","type_order":4,"title":"Disambiguation of Named Entities in Cultural Heritage Texts Using Linked Data Sets","year":2015,"authors":["Brando, C.","Frontini, F.","Ganascia, J."],"authors_source":"Carmen Brando; Francesca Frontini; JeanGabriel Ganascia","authors_cnr_name":["FRONTINI, FRANCESCA"],"authors_cnr_id":["rp02790"],"authors_cnr_institute":[],"abstract":"This paper proposes a graph-based algorithm baptized REDEN for the disambiguation of authors' names in French literary criticism texts and scientific essays from the 19th century. It leverages knowledge from different Linked Data sources in order to select candidates for each author mention, then performs fusion of DBpedia and BnF individuals into a single graph, and finally decides the best referent using the notion of graph centrality. Some experiments are conducted in order to identify the best size of disambiguation context and to assess the influence on centrality of specific relations represented as edges. This work will help scholars to trace the impact of authors' ideas across different works and time periods","keywords":["Named-entity disambiguation Centrality Linked data Data fusion Digital humanities"],"pages":"505-514","url":"http:\/\/link.springer.com\/chapter\/10.1007%2F978-3-319-23201-0_51","volume":"","doi":"10.1007\/978-3-319-23201-0_51","editors":["Morzy, T.","Valduriez, P.","Bellatreche, L."],"editors_source":"Tadeusz Morzy, Patrick Valduriez, Ladjel Bellatreche","published":"New Trends in Databases and Information Systems","publisher":"","issn":"","isbn":"978-3-319-23200-3","conference_name":"","conference_place":"","conference_date":"","last_updated_cnr":"2024-05-12 11:38:03","last_updated_oai":"2024-05-12 11:38:03","last_updated_www":"0000-00-00 00:00:00"},{"id":2339,"id_source":305311,"institutes":["ILC"],"type":"conference_article","type_order":7,"title":"(Re)thinking the BLARK for Ancient Greek","year":2015,"authors":["Boschetti, F.","Del Gratta, R.","Frontini, F.","Khan, F.","Monachini, M."],"authors_source":"Boschetti, Federico; DEL GRATTA, Riccardo; Frontini, Francesca; Khan, Fahad; Monachini, Monica","authors_cnr_name":["BOSCHETTI, FEDERICO","DEL GRATTA, RICCARDO","FRONTINI, FRANCESCA","MONACHINI, MONICA"],"authors_cnr_id":["rp04876","rp00284","rp02790","rp19457"],"authors_cnr_institute":[],"abstract":"The paper discusses the Basic LAnguage Resource Kit (BLARK) for Ancient Greek, measuring the BLARK matrix against what is actually available for this language, and assessing its applicability to ancient languages in general. In addition, the BLARK and the FLaReNet recommendations are used to define priorities in the sector in close collaboration between philologists and the broader LRT community","keywords":[],"pages":"","url":"https:\/\/iris.cnr.it\/handle\/20.500.14243\/305311","volume":"","doi":"","editors":[],"editors_source":"","published":"","publisher":"","issn":"","isbn":"978-83-932640-8-7","conference_name":"","conference_place":"","conference_date":"","last_updated_cnr":"2024-10-02 12:11:41","last_updated_oai":"0000-00-00 00:00:00","last_updated_www":"0000-00-00 00:00:00"},{"id":1938,"id_source":294757,"institutes":["ILC"],"type":"conference_article","type_order":7,"title":"Une mesure d'int\u00e9r\u00eat \u00e0 base de surrepr\u00e9sentation pour l'extraction des motifs syntaxiques stylistiques","year":2015,"authors":["Boukhaled, M.","Frontini, F.","Ganascia, J."],"authors_source":"Boukhaled, Mohamedamine; Frontini, Francesca; Ganascia, Jeangabriel","authors_cnr_name":["FRONTINI, FRANCESCA"],"authors_cnr_id":["rp02790"],"authors_cnr_institute":[],"abstract":"Dans cette contribution, nous pr\u00e9sentons une \u00e9tude sur la stylistique computationnelle des textes de la litt\u00e9rature classiques fran\u00e7aise fond\u00e9e sur une approche conduite par donn\u00e9es, o\u00f9 la d\u00e9couverte des motifs linguistiques int\u00e9ressants se fait sans aucune connaissance pr\u00e9alable. Nous proposons une mesure objective capable de capturer et d'extraire des motifs syntaxiques stylistiques significatifs \u00e0 partir d'un oeuvre d'un auteur donn\u00e9. Notre hypoth\u00e8se de travail est fond\u00e9e sur le fait que les motifs syntaxiques les plus pertinents devraient refl\u00e9ter de mani\u00e8re significative le choix stylistique de l'auteur, et donc ils doivent pr\u00e9senter une sorte de comportement de surrepr\u00e9sentation contr\u00f4l\u00e9 par les objectifs de l'auteur. Les r\u00e9sultats analys\u00e9s montrent l'efficacit\u00e9 dans l'extraction de motifs syntaxiques int\u00e9ressants dans le texte litt\u00e9raire fran\u00e7ais classique, et semblent particuli\u00e8rement prometteurs pour les analyses de ce type particulier de texte","keywords":["Computational stylistic","text mining","syntactic patterns","interestingness measure"],"pages":"391-396","url":"http:\/\/www.atala.org\/taln_archives\/TALN\/TALN-2015\/taln-2015-court-012.html","volume":"","doi":"","editors":[],"editors_source":"","published":"Actes de La 22e Confe\u0301rence Sur Le Traitement Automatique Des Langues Naturelles","publisher":"","issn":"","isbn":"","conference_name":"22e Confe\u0301rence Sur Le Traitement Automatique Des Langues Naturelles (TALN 2015)","conference_place":"","conference_date":"","last_updated_cnr":"2024-04-24 12:22:27","last_updated_oai":"2024-04-24 12:22:27","last_updated_www":"0000-00-00 00:00:00"},{"id":1440,"id_source":297255,"institutes":["ILC"],"type":"conference_article","type_order":7,"title":"A Peculiarity-based Exploration of Syntactical Patterns: a Computational Study of Stylistics","year":2015,"authors":["Boukhaled, M.","Frontini, F.","Ganascia, J."],"authors_source":"Boukhaled, Mohamedamine; Frontini, Francesca; Ganascia, Jeangabriel","authors_cnr_name":["FRONTINI, FRANCESCA"],"authors_cnr_id":["rp02790"],"authors_cnr_institute":[],"abstract":"In this contribution, we present a computational stylistic study and comparison of classic French literary texts based on a datadriven approach where discovering interesting linguistic patterns is done without any prior knowledge. We propose an objective measure capable of capturing and extracting meaningful stylistic syntactic patterns from a given author's work. Our hypothesis is based on the fact that the most relevant syntactic patterns should significantly reflect the author's stylistic choice and thus they should exhibit some kind of peculiar overrepresentation behavior controlled by the author's purpose with respect to a linguistic norm. The analyzed results show the effectiveness in extracting interesting syntactic patterns from novels, and seem particularly promising for the analysis of such particular texts","keywords":["Computational Stylistics","Interestingness Measure","Sequential Pattern Mining","Syntactic Style"],"pages":"31-39","url":"http:\/\/ceur-ws.org\/Vol-1410\/paper5.pdf","volume":"1410","doi":"","editors":[],"editors_source":"","published":"","publisher":"","issn":"","isbn":"","conference_name":"Workshop on Interactions between Data Mining and Natural Language Processing 2015 co-located with European Conference on Machine Learning and Principles and Practice of Knowledge Discovery in Databases (ECML PKDD 2015)","conference_place":"","conference_date":"","last_updated_cnr":"2024-05-26 17:34:38","last_updated_oai":"2024-05-26 17:34:38","last_updated_www":"0000-00-00 00:00:00"},{"id":1360,"id_source":340774,"institutes":["ILC"],"type":"conference_article","type_order":7,"title":"Linked data for toponym linking in French literary texts","year":2015,"authors":["Brando, C.","Frontini, F.","Ganascia, J."],"authors_source":"Carmen Brando; Francesca Frontini; JeanGabriel Ganascia","authors_cnr_name":["FRONTINI, FRANCESCA"],"authors_cnr_id":["rp02790"],"authors_cnr_institute":[],"abstract":"The present article discusses first experiments in toponym linking of Modern French digital editions aiming to provide an external referent to Linked Data sources. We have so far focused on testing two knowledge bases-French DBpedia and Geonames-for recall. Results highlight quality issues in these data sets for usage in NLP-tasks in domain-specific heritage texts","keywords":["Named-Entity Linking","Linked Data","Digital Humanities"],"pages":"","url":"https:\/\/iris.cnr.it\/handle\/20.500.14243\/340774","volume":"","doi":"10.1145\/2837689.2837699","editors":["Purves, R. S.","Jones, C. B."],"editors_source":"Ross S. Purves, Christopher B. Jones","published":"GIR '15 Proceedings of the 9th Workshop on Geographic Information Retrieval","publisher":"","issn":"","isbn":"978-1-4503-3937-7","conference_name":"GIR'15 9th Workshop on Geographic Information Retrieval","conference_place":"","conference_date":"","last_updated_cnr":"2025-01-03 22:21:34","last_updated_oai":"2025-01-03 22:21:34","last_updated_www":"0000-00-00 00:00:00"},{"id":326,"id_source":307390,"institutes":["ILC"],"type":"conference_article","type_order":7,"title":"Visualising Italian Language Resources: a Snapshot","year":2015,"authors":["Del Gratta, R.","Frontini, F.","Monachini, M.","Pardelli, G.","Russo, I.","Bartolini, R.","Goggi, S.","Khan, F.","Quochi, V.","Soria, C.","Calzolari, N."],"authors_source":"DEL GRATTA, Riccardo; Frontini, Francesca; Monachini, Monica; Pardelli, Gabriella; Russo, Irene; Bartolini, Roberto; Goggi, Sara; Khan, Fahad; Quochi, Valeria; Soria, Claudia; Calzolari, Nicoletta","authors_cnr_name":["DEL GRATTA, RICCARDO","FRONTINI, FRANCESCA","MONACHINI, MONICA","PARDELLI, GABRIELLA","RUSSO, IRENE","BARTOLINI, ROBERTO","GOGGI, SARA","QUOCHI, VALERIA","SORIA, CLAUDIA"],"authors_cnr_id":["rp00284","rp02790","rp19457","rp20820","rp02389","rp00239","rp20855","rp13283","rp17652"],"authors_cnr_institute":[],"abstract":"This paper aims to provide a first snapshot of Italian Language Resources (LRs) and their uses by the community, as documented by the papers presented at two different conferences, LREC2014 and CLiC-it 2014. The data of the former were drawn from the LOD version of the LRE Map, while those of the latter come from manually analyzing the proceedings. The results are presented in the form of visual graphs and confirm the initial hypothesis that Italian LRs require concrete actions to enhance their visibility","keywords":["Italian Language Resources"],"pages":"100-104","url":"https:\/\/books.openedition.org\/aaccademia\/1277?lang=it","volume":"","doi":"","editors":["Bosco, C.","Tonelli, S.","Zanzotto, F. M."],"editors_source":"Cristina Bosco, Sara Tonelli, Fabio Massimo Zanzotto","published":"Proceedings of the Second Italian Conference on Computational Linguistics CLiC-it 2015","publisher":"","issn":"","isbn":"978-88-99200-62-6","conference_name":"Second Italian Conference on Computational Linguistics CLiC-it 2015","conference_place":"","conference_date":"","last_updated_cnr":"2024-10-02 12:08:04","last_updated_oai":"2024-10-02 12:08:04","last_updated_www":"0000-00-00 00:00:00"},{"id":1189,"id_source":276113,"institutes":["ILC"],"type":"conference_article","type_order":7,"title":"Linguistic Pattern Extraction and Analysis for Classic French Plays","year":2015,"authors":["Frontini, F.","Amine Boukhaled, M.","Ganascia, J."],"authors_source":"Frontini, Francesca; Amine Boukhaled, Mohamed; Ganascia, Jeangabriel","authors_cnr_name":["FRONTINI, FRANCESCA"],"authors_cnr_id":["rp02790"],"authors_cnr_institute":[],"abstract":"Great authors of fiction and theatre have the capacity of creating memorable characters that take life and become almost as real as living persons to the readers\/audience. The study of characterization, namely of how this is achieved, is a well-researched topic in corpus stylistics: for instance (Mahlberg, 2012) attempts to identify typical lexical patterns for memorable Dickens' characters by extracting those lexical bundles that stand out (namely are overrepresented) in comparison to a general corpus. In other works, authorship attribution methods are applied to the different characters of a play to identify whether the author has been able to provide each of them with a \"distinct\" voice. For instance (Vogel & Lynch, 2008) compare individual Shakespeare characters against the whole play or even against all plays of the same author. The purpose of this paper is to propose a methodology for the study characterization of several characters in French plays of the classical period. The tools developed are meant to support textual analysis by: 1) Verifying the degree of characterization of each character with respect to others. 2) Automatically inducing a list of linguistic features that are significant, representative for that character. Preliminary investigations have been conducted on plays by Moliere, cross-comparing four protagonists from four different plays. The proposed methodology relies on sequential data mining for the extraction of linguistic patterns and on correspondence analysis for comparison of patterns frequencies in each character and for the visual representation of such differences","keywords":["computational stylometry","thater","sequential pattern mining"],"pages":"3","url":"http:\/\/lipn.univ-paris13.fr\/~charnois\/conscilaGenres\/resumes\/frontini.pdf","volume":"","doi":"","editors":[],"editors_source":"","published":"","publisher":"","issn":"","isbn":"","conference_name":"Journe\u0301e ConSciLa (Confrontations en Sciences du Langage) Grammaire des genres et des styles: quelles approches privile\u0301gier ?","conference_place":"","conference_date":"","last_updated_cnr":"2024-05-07 11:03:53","last_updated_oai":"2024-05-07 11:03:53","last_updated_www":"0000-00-00 00:00:00"},{"id":1088,"id_source":290872,"institutes":["ILC"],"type":"conference_article","type_order":7,"title":"Semantic Web based Named Entity Linking for Digital Humanities and Heritage Texts","year":2015,"authors":["Frontini, F.","Brando, C.","Ganascia, J."],"authors_source":"Francesca Frontini; Carmen Brando; JeanGabriel Ganascia","authors_cnr_name":["FRONTINI, FRANCESCA"],"authors_cnr_id":["rp02790"],"authors_cnr_institute":[],"abstract":"This paper proposes a graph based methodology for automatically disambiguating authors' mentions in a corpus of French literary criticism. Candidate referents are identified and evaluated using a graph based named entity linking algorithm, which exploits a knowledge-base built out of two different resources (DBpedia and the BnF linked data). The algorithm expands previous ones applied for word sense disambiguation and entity linking, with good results. Its novelty resides in the fact that it successfully combines a generic knowledge base such as DBpedia with a domain specific one, thus enabling the efficient annotation of minor authors. This will help specialists to follow mentions of the same author in different works of literary criticism, and thus to investigate their literary appreciation over time","keywords":["named-entity linking","linked data","digital humanities"],"pages":"77-88","url":"http:\/\/ceur-ws.org\/Vol-1364\/paper9.pdf","volume":"VOL-1364","doi":"","editors":["Zucker, A.","Draelants, I.","Zucker, C. F.","Monnin, A."],"editors_source":"Arnaud Zucker , Isabelle Draelants , Catherine Faron Zucker , Alexandre Monnin","published":"SW4SH 2015 Semantic Web for Scientific Heritage 2015","publisher":"","issn":"","isbn":"","conference_name":"SW4SH 2015 Semantic Web for Scientific Heritage 2015","conference_place":"","conference_date":"","last_updated_cnr":"2024-06-18 15:31:02","last_updated_oai":"2024-06-18 15:31:02","last_updated_www":"0000-00-00 00:00:00"},{"id":646,"id_source":295464,"institutes":["ILC"],"type":"conference_article","type_order":7,"title":"Domain-adapted named-entity linker using Linked Data","year":2015,"authors":["Frontini, F.","Brando, C.","Ganascia, J."],"authors_source":"Francesca Frontini; Carmen Brando; JeanGabriel Ganascia","authors_cnr_name":["FRONTINI, FRANCESCA"],"authors_cnr_id":["rp02790"],"authors_cnr_institute":[],"abstract":"We present REDEN, a tool for graph-based Named Entity Linking that allows for the disambiguation of entities using domain-specific Linked Data sources and different configurations (e. g. context size). It takes TEI-annotated texts as input and outputs them enriched with external references (URIs). The possibility of customizing indexes built from various knowledge sources by defining temporal and spatial extents makes REDEN particularly suited to handle domain-specific corpora such as enriched digital editions in the Digital Humanities","keywords":["named-entity disambiguation","evaluation","linked data","digital humanities"],"pages":"10","url":"http:\/\/ceur-ws.org\/Vol-1386\/named_entity.pdf","volume":"VOL-1386","doi":"","editors":["Izquierdo, R."],"editors_source":"Ruben Izquierdo","published":"Proceedings of the Workshop on NLP Applications: Completing the Puzzle","publisher":"","issn":"","isbn":"","conference_name":"Workshop on NLP Applications: Completing the Puzzle co-located with the 20th International Conference on Applications of Natural Language to Information Systems (NLDB 2015)","conference_place":"","conference_date":"","last_updated_cnr":"2024-04-24 11:32:15","last_updated_oai":"2024-04-24 11:32:15","last_updated_www":"0000-00-00 00:00:00"},{"id":146,"id_source":267184,"institutes":["ILC"],"type":"conference_article","type_order":7,"title":"Generative Lexicon and polysemy: inducing logical alternations","year":2015,"authors":["Frontini, F.","Quochi, V.","Monachini, M."],"authors_source":"Frontini, Francesca; Quochi, Valeria; Monachini, Monica","authors_cnr_name":["FRONTINI, FRANCESCA","QUOCHI, VALERIA","MONACHINI, MONICA"],"authors_cnr_id":["rp02790","rp13283","rp19457"],"authors_cnr_institute":[],"abstract":"The current paper brings together the results of a series of experiments for inducing regular sense alternations, or regular\/ logical polysemy, from a computational lexicon based on the Generative Lexicon theory. The results are discussed in light of the potential benefits and uses of the amended algorithm","keywords":["Polysemy","Generative Lexicon","Logical Alternations"],"pages":"7","url":"https:\/\/iris.cnr.it\/handle\/20.500.14243\/267184","volume":"","doi":"","editors":["Hsieh, S. K.","Kanzaki, K."],"editors_source":"Shu-Kai Hsieh and Kyoko Kanzaki (eds.)","published":"","publisher":"MAPLEX2015 Multiple Approaches to Lexicon Conference (Yamagata, JPN)","issn":"","isbn":"","conference_name":"MAPLEX2015 Multiple Approaches to Lexicon Conference","conference_place":"Yamagata","conference_date":"","last_updated_cnr":"2024-06-16 12:53:39","last_updated_oai":"2024-06-16 12:53:39","last_updated_www":"0000-00-00 00:00:00"},{"id":216,"id_source":290971,"institutes":["ILC"],"type":"conference_article","type_order":7,"title":"Marine Planning and Service Platform (MAPS): An Advanced Research Engine for Grey Literature in Marine Science","year":2015,"authors":["Goggi, S.","Monachini, M.","Frontini, F.","Bartolini, R.","Pardelli, G.","De Mattei, M.","Bustaffa, F.","Manzella, G."],"authors_source":"Goggi, Sara; Monachini, Monica; Frontini, Francesca; Bartolini, Roberto; Pardelli, Gabriella; De Mattei, Maurizio; Bustaffa, Franco; Manzella, Giuseppe","authors_cnr_name":["GOGGI, SARA","MONACHINI, MONICA","FRONTINI, FRANCESCA","BARTOLINI, ROBERTO","PARDELLI, GABRIELLA"],"authors_cnr_id":["rp20855","rp19457","rp02790","rp00239","rp20820"],"authors_cnr_institute":[],"abstract":"The MAPS {Marine Planning and Service Platform} project is a development of the Marine project {Ricerca Industriale e Sviluppo Sperimentale Regione Liguria 2007-2013} aiming at building a computer platform for supporting a Marine Information and Knowledge System, as part of the data management activities. One of the main objective of the project is to develop a repository that should gather, classify and structure marine scientific literature and data thus guaranteeing their accessibility to researchers and institutions by means of standard protocols. We will present the scenario of the Operative Oceanography together with the technologies used to develop an advanced search engine which aims at providing rapid and efficient access to a Digital Library of oceanographic data. The case-study is also highlighting how the retrieval of grey literature from this specific marine community could be reproduced for similar communities as well, thus revealing the great impact that the processing, re-use as well as application of grey data have on societal needs\/problems and their answers","keywords":["Marine Science","Search Engine","Source Data","Oceanography"],"pages":"108-114","url":"http:\/\/www.textrelease.com\/gl16program.html","volume":"","doi":"","editors":["Farace, D.","Frantzen, J."],"editors_source":"D. Farace and J. Frantzen","published":"THE GL-CONFERENCE SERIES. CONFERENCE PROCEEDINGS","publisher":"TextRelease (Amsterdam, NLD)","issn":"1386-2316","isbn":"978-90-77484-23-4","conference_name":"Sixteenth International Conference on Grey Literature Grey Literature Lobby: Engines and Requesters for Change","conference_place":"Amsterdam","conference_date":"","last_updated_cnr":"2024-05-31 11:49:37","last_updated_oai":"2024-05-31 11:49:37","last_updated_www":"0000-00-00 00:00:00"},{"id":661,"id_source":295959,"institutes":["ILC"],"type":"conference_article","type_order":7,"title":"Using Ontologies to Model Polysemy in Lexical Resources","year":2015,"authors":["Khan, F.","Frontini, F."],"authors_source":"Khan, Fahad; Frontini, Francesca","authors_cnr_name":["FRONTINI, FRANCESCA"],"authors_cnr_id":["rp02790"],"authors_cnr_institute":[],"abstract":"In this article we look at how the use of ontologies can assist in analysing polysemy in natural languages. We develop a model, the Lexical-Sense-Ontology model (LSO), to represent the interaction between a lexicon and ontology, based on lemon. We use the LSO model to show how default rules can be used to represent semi-productivity in polysemy as well as discussing the kinds of ontological information that are useful for studying polysemy","keywords":["Polysemy","Ontology","Default Logic"],"pages":"","url":"http:\/\/www.aclweb.org\/anthology\/W\/W15\/W15-0404.pdf","volume":"","doi":"","editors":[],"editors_source":"","published":"Proceedings of the Workshop on Language and Ontologies","publisher":"","issn":"","isbn":"","conference_name":"Workshop on Language and Ontologies","conference_place":"","conference_date":"","last_updated_cnr":"2024-04-07 10:14:22","last_updated_oai":"2024-04-07 10:14:22","last_updated_www":"0000-00-00 00:00:00"},{"id":2344,"id_source":305309,"institutes":["ILC"],"type":"conference_misc","type_order":8,"title":"Strumenti, Risorse e Linguistic Linked Open Data per le lingue antiche","year":2015,"authors":["Boschetti, F.","Del Gratta, R.","Frontini, F.","Khan, A. F.","Monachini, M."],"authors_source":"Federico Boschetti; Riccardo Del Gratta; Francesca Frontini; Anas Fahad Khan; Monica Monachini","authors_cnr_name":["BOSCHETTI, FEDERICO","DEL GRATTA, RICCARDO","FRONTINI, FRANCESCA","MONACHINI, MONICA"],"authors_cnr_id":["rp04876","rp00284","rp02790","rp19457"],"authors_cnr_institute":[],"abstract":"Strumenti e metodi dell'Informatica Umanistica hanno portato e portano ad una ridefinizione di processi teorici, metodologici e tecnici, fino a una vera e propria ri-concettualizzazione dei saperi nell'ambito dei beni culturali. L'Istituto di Linguistica Computazionale \u00e8 attivo con varie iniziative sul fronte delle Digital Humanities per la creazione di strumenti e risorse linguistiche per il mondo classico. La direzione intrapresa si inserisce nel paradigma che si va consolidando nel settore delle tecnologie del linguaggio e che prevede la fruizione di servizi linguistici attraverso infrastrutture di ricerca, secondo un modello gi\u00e0 operativo per le lingue moderne. Tale paradigma \u00e8 in connessione con l'emergere degli standard e dei formati del web semantico per le tecnologie del linguaggio e per la pubblicazione di dati linguistici","keywords":[],"pages":"","url":"https:\/\/iris.cnr.it\/handle\/20.500.14243\/305309","volume":"","doi":"","editors":[],"editors_source":"","published":"","publisher":"","issn":"","isbn":"","conference_name":"","conference_place":"","conference_date":"","last_updated_cnr":"2024-10-03 09:04:22","last_updated_oai":"0000-00-00 00:00:00","last_updated_www":"0000-00-00 00:00:00"},{"id":1662,"id_source":289592,"institutes":["ILC"],"type":"conference_misc","type_order":8,"title":"Moliere's Raisonneurs: a quantitative study of distinctive linguistic patterns","year":2015,"authors":["Frontini, F.","Boukhaled, M. A.","Ganascia, J. G."],"authors_source":"Francesca Frontini; Mohamed Amine Boukhaled; Jean Gabriel Ganascia","authors_cnr_name":["FRONTINI, FRANCESCA"],"authors_cnr_id":["rp02790"],"authors_cnr_institute":[],"abstract":"","keywords":["Computational Stylistics","Correspondence analysis","Corpus linguistics","Molie\u0300re"],"pages":"114-117","url":"http:\/\/ucrel.lancs.ac.uk\/cl2015\/doc\/CL2015-AbstractBook.pdf","volume":"","doi":"","editors":["Formato, F.","Hardie, A."],"editors_source":"Federica Formato and Andrew Hardie","published":"Corpus Linguistics 2015-Abstract Book","publisher":"","issn":"","isbn":"","conference_name":"Corpus Linguistics 2015","conference_place":"","conference_date":"","last_updated_cnr":"2024-05-11 23:56:50","last_updated_oai":"2024-05-11 23:56:50","last_updated_www":"0000-00-00 00:00:00"},{"id":1080,"id_source":307398,"institutes":["ILC"],"type":"conference_misc","type_order":8,"title":"A semantic engine for grey literature retrieval in the oceanography domain","year":2015,"authors":["Goggi, S.","Pardelli, G.","Bartolini, R.","Frontini, F.","Monachini, M.","Manzella, G.","De Mattei, M.","Bustaffa, F."],"authors_source":"Goggi, Sara; Pardelli, Gabriella; Bartolini, Roberto; Frontini, Francesca; Monachini, Monica; Manzella, Giuseppe; De Mattei, Maurizio; Bustaffa, Franco","authors_cnr_name":["GOGGI, SARA","PARDELLI, GABRIELLA","BARTOLINI, ROBERTO","FRONTINI, FRANCESCA","MONACHINI, MONICA"],"authors_cnr_id":["rp20855","rp20820","rp00239","rp02790","rp19457"],"authors_cnr_institute":[],"abstract":"Here we present the final results of MAPS (Marine Planning and Service Platform), an environment designed for gathering, classifying, managing and accessing marine scientific literature and data, making it available for search to Operative Oceanography researchers of various institutions by means of standard protocols. In previous publications the general architecture of the system as well as the set of metadata (Common Data Index) used to describe the documents were presented [3]; it was shown how individual oceanographic data-sets could be indexed within the MAPS library by types of measure, measurement tools, geographic areas, and also linked to specific textual documentation. Documentation is described using the current international standards: Title, Authors, Publisher, Language, Date of publication, Body\/Institution, Abstract, etc.; serial publications are described in terms of ISSN, while books are assigned ISBN; content of various types on electronic networks is described by means of doi and url. Each description is linked to the document. Thanks to this, the MAPS library already enables researchers to go from structured oceanographic data to documents describing it. But this was not enough: documents may contain important information that has not been encoded in the metadata. Thus an advanced Search Engine was put in place that uses semantic-conceptual technologies in order to extract key concepts from unstructured text such as technical documents (reports and grey literature) and scientific papers and to make them indexable and searchable by the end user in the same way as the structured data (such as oceanographic observations and metadata) is. More specifically once a document is uploaded in the MAPS library, key domain concepts in documents are extracted via a natural language processing pipeline and used as additional information for its indexing. The key term identification algorithm is based on marine concepts that were pre-defined in a domain ontology, but crucially it also allows for the discovery of new related concepts. So for instance starting from the domain term salinity, related terms such as sea salinity and average sea salinity will also be identified as key terms and used for indexing and searching documents. A hybrid search system is then put in place, where users can search the library by metadata or by free text queries. In the latter case, the NLP pipeline performs an analysis of the text of the query, and when key concepts are matched, the relevant documents are presented. The results may be later refined by using other structured information (e. g. date of publication, area,.). Currently a running system has been put in place, with data from satellites, buoys and sea stations; such data is documented and searchable by its relevant metadata and documentation. Results of quantitative evaluation in terms of information retrieval measures will be presented in the poster; more specifically, given an evaluation set defined by domain experts and composed of pre-defined queries together with documents that answer such queries, it will be shown how the system is highly accurate in retrieving the correct documents from the library. Though this work focuses on oceanography, its results may be easily extended to other domains; more generally, the possibility of enhancing the visibility and accessibility of grey literature via its connection to the data it describes and to an advanced full text indexing are of great relevance for the topic of this conference","keywords":["Information Extraction","Search Engine","Oceanography"],"pages":"76-77","url":"https:\/\/iris.cnr.it\/handle\/20.500.14243\/307398","volume":"","doi":"","editors":["Farace, D.","Frantzen, J."],"editors_source":"Dominic Farace, Jerry Frantzen","published":"GL17 Program Book","publisher":"","issn":"","isbn":"978-90-77484-26-5","conference_name":"Seventeenth International Conference on Grey Literature. A New Wave of Textual and Non-Textual Grey Literature","conference_place":"","conference_date":"","last_updated_cnr":"2024-05-04 11:51:34","last_updated_oai":"2024-05-04 11:51:34","last_updated_www":"0000-00-00 00:00:00"},{"id":443,"id_source":300554,"institutes":["ILC"],"type":"misc","type_order":11,"title":"Reconnaissance d'entit\u00e9s nomm\u00e9es: adaptation au domaine de la litt\u00e9rature fran\u00e7aise du XIXe si\u00e8cle","year":2015,"authors":["Brando, C.","Frontini, F.","Abi Haidar, A.","Ganascia, J."],"authors_source":"Brando, Carmen; Frontini, Francesca; Abi Haidar, Alaa; Ganascia, Jeangabriel","authors_cnr_name":["FRONTINI, FRANCESCA"],"authors_cnr_id":["rp02790"],"authors_cnr_institute":[],"abstract":"La reconnaissance d'entit\u00e9s nomm\u00e9es (REN) est un enjeu fondamental pour la recherche en humanit\u00e9s num\u00e9riques (HN). En litt\u00e9rature fran\u00e7aise, il est particuli\u00e8rement important de rep\u00e9rer des entit\u00e9s telles que les auteurs, les personnages fictifs, les lieux g\u00e9ographiques et imaginaires, les titres d'ouvrages, les marqueurs temporels, entre autres. Actuellement, il existe peu de corpus de litt\u00e9rature fran\u00e7aise du pass\u00e9 annot\u00e9s et disponibles en ligne. Le co\u00fbt \u00e9lev\u00e9 de l'annotation manuelle motive donc l'utilisation de m\u00e9thodes automatiques. Les approches REN de l'\u00e9tat de l'art fonctionnent efficacement sur des corpus journalistique et de litt\u00e9rature scientifique en biologie [1]. N\u00e9anmoins, l'adaptation \u00e0 un nouveau domaine semble affecter n\u00e9gativement la performance de ces approches [5]. La diversit\u00e9 des textes en litt\u00e9rature (fiction, critique, th\u00e9\u00e2tre.) et la sp\u00e9cificit\u00e9 des \u00e9poques prises en compte repr\u00e9sentent un travail consid\u00e9rable d'adaptation des ressources linguistiques et des algorithmes \u00e0 un domaine particulier. En g\u00e9n\u00e9ral, les th\u00e8mes trait\u00e9s sont h\u00e9t\u00e9rog\u00e8nes et les textes poss\u00e8dent un style fr\u00e9quemment caract\u00e9ris\u00e9 par un bas degr\u00e9 de standardisation et de pr\u00e9dictibilit\u00e9. Il est par exemple difficile d'identifier des mentions candidates car les conventions typographiques et le registre linguistique varient selon le domaine (textes journalistiques vs. litt\u00e9rature fran\u00e7aise)","keywords":["entite\u0301s nommee\u0301s","litte\u0301rature franc\u0327aise"],"pages":"","url":"https:\/\/iris.cnr.it\/handle\/20.500.14243\/300554","volume":"","doi":"","editors":[],"editors_source":"","published":"","publisher":"","issn":"","isbn":"","conference_name":"8esJourne\u0301es Internationales de Linguistique de Corpus (JLC2015)","conference_place":"","conference_date":"","last_updated_cnr":"2024-04-06 11:42:34","last_updated_oai":"2024-04-06 11:42:34","last_updated_www":"0000-00-00 00:00:00"},{"id":670,"id_source":300594,"institutes":["ILC"],"type":"misc","type_order":11,"title":"Trattamento automatico del linguaggio per le Digital Humanities. Riconoscimento e disambiguazione di menzioni di autori in testi di critica letteraria","year":2015,"authors":["Frontini, F."],"authors_source":"Francesca Frontini","authors_cnr_name":["FRONTINI, FRANCESCA"],"authors_cnr_id":["rp02790"],"authors_cnr_institute":[],"abstract":"L'intervento scaturisce da una collaborazione tra ILC-CNR e il Labex OBVIL di Parigi. Lo scopo del progetto \u00e8 quello di adattare ed estendere algoritmi di riconoscimento, classificazione e disambiguazione di entit\u00e0 nominate (in particolare menzioni di autori) nel \"Corpus Critique\", un insieme di testi di critica letteraria francese che il Labex OBVIL sta pubblicando in edizione digitale (formato TEI). Tali algoritmi si basano su approcci TAL supervisionati e non supervisionati e sfruttano massicciamente le basi di conoscenza, sia generiche (DBpedia) che di dominio, disponibili online sotto forma di linked data; lo scopo di tali lavori \u00e8 di produrre risorse testuali annotate per facilitare la ricerca nell'ambito della storia della critica letteraria e della storia delle idee in generale. Durante il seminario verranno introdotti i formati e le risorse utilizzate, i criteri e le problematiche di annotazione emersi, e gli algoritmi riconoscimento e disambiguazione di entit\u00e0 nominate sviluppati. Pi\u00f9 in generale si cercher\u00e0 di mostrare con alcuni casi di utilizzo quali siano i vantaggi di arricchire risorse testuali con questo livello di annotazione, nel pi\u00f9 ampio contesto delle convergenze tra digital humanities e trattamento automatico del linguaggio. Link http: \/\/obvil. paris-sorbonne. fr\/ https: \/\/github. com\/cvbrandoe\/REDEN\/blob\/master\/README. md","keywords":["Named-entity disambiguation Centrality Linked data Data fusion Digital humanities"],"pages":"","url":"https:\/\/iris.cnr.it\/handle\/20.500.14243\/300594","volume":"","doi":"","editors":[],"editors_source":"","published":"","publisher":"","issn":"","isbn":"","conference_name":"Seminario di Cultura Digitale","conference_place":"","conference_date":"","last_updated_cnr":"2024-05-15 12:27:31","last_updated_oai":"2024-05-15 12:27:31","last_updated_www":"0000-00-00 00:00:00"},{"id":913,"id_source":295960,"institutes":["ILC"],"type":"misc","type_order":11,"title":"Analyse et extraction des motifs syntaxiques dans la prose de Robert Challe et de ses apocryphes","year":2015,"authors":["Frontini, F."],"authors_source":"Francesca Frontini","authors_cnr_name":["FRONTINI, FRANCESCA"],"authors_cnr_id":["rp02790"],"authors_cnr_institute":[],"abstract":"Cette contribution presente une extraction et une analyse des motifs syntaxiques dans la prose de Robert Challe et de ses apocryphes. En particulier nous analysons les diff\u00e9rence dans la syntaxe des contes originaux des Illustres Fran\u00e7aises et celle des contes apocryphes","keywords":["Robert Challe","authorship attribution","stilistica computazionale"],"pages":"","url":"http:\/\/obvil.paris-sorbonne.fr\/sites\/default\/files\/projets\/analyse_motifs_syntaxiques_if_et_apocryphes.pdf","volume":"","doi":"","editors":[],"editors_source":"","published":"","publisher":"","issn":"","isbn":"","conference_name":"Robert Challe: approches nume\u0301riques des questions d'auctorialite\u0301","conference_place":"","conference_date":"","last_updated_cnr":"2024-03-20 04:45:55","last_updated_oai":"2024-03-20 04:45:55","last_updated_www":"0000-00-00 00:00:00"},{"id":1892,"id_source":289092,"institutes":["ILC"],"type":"misc","type_order":11,"title":"What makes them different: the extraction of distinctive linguistic patterns for the protagonists of Moli\u00e8re's plays","year":2015,"authors":["Frontini, F."],"authors_source":"Francesca Frontini","authors_cnr_name":["FRONTINI, FRANCESCA"],"authors_cnr_id":["rp02790"],"authors_cnr_institute":[],"abstract":"Quantitative approaches to the study of style in literature are far from a modern novelty. They have however recently gained more and more popularity, not only among computer scientists and corpus linguistics, but also among some influential literary critics. The present panorama of quantitative techniques is very rich, but often confusing, with a plethora of denominations and methodologies often difficult to reconcile; computer scientists classify their work as stylometry or computational stylistics, while linguists may use the label corpus stylistics, and finally critics like Franco Moretti will talk about macro-analysis and distant reading. This talk will try first to identify the differences between these trends, distinguishing between corpus based and corpus driven approaches on the methodological side (Quiniou et al 2012), and (following Ramsey 2011) between experimental and hermeneutical approaches. Finally we will present ongoing work conducted at Labex OBVIL on syntactic pattern extraction from theatrical characters. The proposed approach, using correspondence analysis to extract distinctive traits for each character, is imagined rather as an hermeneutical tool, in the sense that it does not seek to demonstrate that two different characters have been endowed with significantly different stylistic traits by the playwright, but it does enable the visualisation of their relative distances and the extraction of those elements that make them distinct","keywords":[],"pages":"","url":"https:\/\/iris.cnr.it\/handle\/20.500.14243\/289092","volume":"","doi":"","editors":[],"editors_source":"","published":"","publisher":"","issn":"","isbn":"","conference_name":"Cycle des se\u0301minaires ILES LIMSI","conference_place":"","conference_date":"","last_updated_cnr":"2024-06-11 13:06:54","last_updated_oai":"2024-06-11 13:06:54","last_updated_www":"0000-00-00 00:00:00"},{"id":2336,"id_source":295465,"institutes":["ILC"],"type":"misc","type_order":11,"title":"Indexing names in digital editions","year":2015,"authors":["Frontini, F."],"authors_source":"Francesca Frontini","authors_cnr_name":["FRONTINI, FRANCESCA"],"authors_cnr_id":["rp02790"],"authors_cnr_institute":[],"abstract":"This presentation outlines the work done on Named Entity Recognition and Linking in texts of French Literary criticism, underlying the points of interest for what concerns the creation of enriched digital editions","keywords":[],"pages":"","url":"https:\/\/iris.cnr.it\/handle\/20.500.14243\/295465","volume":"","doi":"","editors":[],"editors_source":"","published":"","publisher":"","issn":"","isbn":"","conference_name":"","conference_place":"","conference_date":"","last_updated_cnr":"2024-06-25 17:18:24","last_updated_oai":"0000-00-00 00:00:00","last_updated_www":"0000-00-00 00:00:00"},{"id":2072,"id_source":296549,"institutes":["ILC"],"type":"misc","type_order":11,"title":"Mining for characterising patterns in literature using correspondence analysis: an experiment on French novels","year":2015,"authors":["Frontini, F."],"authors_source":"Francesca Frontini","authors_cnr_name":["FRONTINI, FRANCESCA"],"authors_cnr_id":["rp02790"],"authors_cnr_institute":[],"abstract":"The talk presents and describes a bottom up methodology for the detection of stylistic traits in the syntax of literary texts. The extraction of syntactic patterns is performed blindly by a sequential pattern mining algorithm, while the identification of significant and interesting features is performed later by using correspondence analysis and filtering for the most contributive patterns","keywords":["computational stylistics","French"],"pages":"","url":"https:\/\/iris.cnr.it\/handle\/20.500.14243\/296549","volume":"","doi":"","editors":[],"editors_source":"","published":"","publisher":"","issn":"","isbn":"","conference_name":"Go\u0308ttingen Dialog in Digital Humanities","conference_place":"","conference_date":"","last_updated_cnr":"2024-06-18 14:53:43","last_updated_oai":"2024-06-18 14:53:43","last_updated_www":"0000-00-00 00:00:00"},{"id":1181,"id_source":257904,"institutes":["ILC"],"type":"conference_article","type_order":7,"title":"The LREMap for Under-Resourced Languages","year":2014,"authors":["Del Gratta, R.","Frontini, F.","Khan, F.","Mariani, J.","Soria, C."],"authors_source":"DEL GRATTA, Riccardo; Frontini, Francesca; Khan, Fahad; Mariani, Joseph; Soria, Claudia","authors_cnr_name":["DEL GRATTA, RICCARDO","FRONTINI, FRANCESCA","SORIA, CLAUDIA"],"authors_cnr_id":["rp00284","rp02790","rp17652"],"authors_cnr_institute":[],"abstract":"A complete picture of currently available language resources and technologies for the under-resourced languages of Europe is still lacking. Yet this would help policy makers, researchers and developers enormously in planning a roadmap for providing all languages with the necessary instruments to act as fully equipped languages in the digital era. In this paper we introduce the LRE Map and show its utility for documenting available language resources and technologies for under-resourced languages. The importance of the serialization of the LREMap into (L)LOD along with the possibility of its connection to a wider world is also introduced","keywords":["language resources","less-resourced languages","linguistic linked open data"],"pages":"78-83","url":"http:\/\/www.lrec-conf.org\/proceedings\/lrec2014\/index.html","volume":"","doi":"","editors":["Pretorius, L.","Soria, C.","Baroni, P."],"editors_source":"Laurette Pretorius, Claudia Soria, Paola Baroni","published":"Proceedings of the Workshop on Collaboration and Computing for Under-Resourced Languages in the Linked Open Data Era (CCURL 2014)","publisher":"","issn":"","isbn":"","conference_name":"Workshop on Collaboration and Computing for Under-Resourced Languages in the Linked Open Data Era (CCURL 2014)","conference_place":"","conference_date":"","last_updated_cnr":"2024-10-02 12:41:48","last_updated_oai":"2024-10-02 12:41:48","last_updated_www":"0000-00-00 00:00:00"},{"id":609,"id_source":259129,"institutes":["ILC"],"type":"conference_article","type_order":7,"title":"Polysemy alternations extraction using the PAROLE SIMPLE CLIPS Italian lexicon","year":2014,"authors":["Frontini, F.","Quochi, V.","Monachini, M."],"authors_source":"Frontini F.; Quochi V.; Monachini M.","authors_cnr_name":["FRONTINI, FRANCESCA","QUOCHI, VALERIA","MONACHINI, MONICA"],"authors_cnr_id":["rp02790","rp13283","rp19457"],"authors_cnr_institute":[],"abstract":"This paper presents the results of an experiment of polysemy alternations induction from a lexicon (Utt and Pad\u00b4o, 2011; Frontini et al., 2014), discussing the results and proposing an amendment in the original algorithm","keywords":["Language Resources and Technologies"],"pages":"175-179","url":"http:\/\/clic.humnet.unipi.it\/proceedings\/Proceedings-CLICit-2014.pdf","volume":"","doi":"10.12871\/CLICIT2014134","editors":["Basili, R.","Lenci, A.","Magnini, B."],"editors_source":"Roberto Basili, Alessandro Lenci, Bernardo Magnini","published":"","publisher":"Pisa University Press srl (Pisa, ITA)","issn":"","isbn":"978-88-67-41472-7","conference_name":"Proceedings of the First Italian Conference on Computational Linguistics CLiC-it 2014 & the Fourth International Workshop EVALITA 2014","conference_place":"Pisa","conference_date":"","last_updated_cnr":"2024-06-18 15:28:25","last_updated_oai":"2024-06-18 15:28:25","last_updated_www":"0000-00-00 00:00:00"},{"id":392,"id_source":222781,"institutes":["ILC"],"type":"conference_article","type_order":7,"title":"Polysemy Index for Nouns: an Experiment on Italian using the PAROLE SIMPLE CLIPS Lexical Database","year":2014,"authors":["Frontini, F.","Quochi, V.","Pad\u00f3, S.","Utt, J.","Monachini, M."],"authors_source":"Frontini Francesca; Valeria Quochi; Sebastian Pad\u00f3; Jason Utt; Monica Monachini","authors_cnr_name":["FRONTINI, FRANCESCA","QUOCHI, VALERIA","MONACHINI, MONICA"],"authors_cnr_id":["rp02790","rp13283","rp19457"],"authors_cnr_institute":[],"abstract":"An experiment is presented to induce a set of polysemous basic type alternations (such as ANIMAL-FOOD, or BUILDING-INSTITUTION) by deriving them from the sense alternations found in an existing lexical resource. The paper builds on previous work and applies those results to the Italian lexicon PAROLE SIMPLE CLIPS. The new results show how the set of frequent type alternations that can be induced from the lexicon is partly different from the set of polysemy relations selected and explicitly applied by lexicographers when building it. The analysis of mismatches shows that frequent type alternations do not always correspond to prototypical polysemy relations, nevertheless the proposed methodology represents a useful tool offered to lexicographers to systematically check for possible gaps in their resource","keywords":["Polysemy","lexical resources","semantics"],"pages":"2955-2963","url":"http:\/\/www.lrec-conf.org\/proceedings\/lrec2014\/index.html","volume":"","doi":"","editors":["Calzolari, N.","Choukri, K.","Declerck, T.","Loftsson, H.","Maegaard, B.","Mariani, J.","Moreno, A.","Odijk, J.","Piperidis, S."],"editors_source":"N. Calzolari, K. Choukri, T. Declerck, H. Loftsson, B. Maegaard, J. Mariani, A. Moreno, J. Odijk, S. Piperidis","published":"LREC 2014 Ninth International Conference on Language Resources and Evaluation Proceedings","publisher":"European Language Resources Association ELRA (Paris, FRA)","issn":"","isbn":"978-2-9517408-8-4","conference_name":"9th International Conference on Language Resources and Evaluation, LREC 2014","conference_place":"Paris","conference_date":"","last_updated_cnr":"2024-04-23 23:55:45","last_updated_oai":"2024-04-23 23:55:45","last_updated_www":"0000-00-00 00:00:00"},{"id":2396,"id_source":259370,"institutes":["ILC"],"type":"conference_article","type_order":7,"title":"Using lemon to Model Lexical Semantic \u00a0Shift in Diachronic Lexical Resources","year":2014,"authors":["Khan, F.","Boschetti, F.","Frontini, F."],"authors_source":"Fahad Khan; Federico Boschetti; Francesca Frontini","authors_cnr_name":["BOSCHETTI, FEDERICO","FRONTINI, FRANCESCA"],"authors_cnr_id":["rp04876","rp02790"],"authors_cnr_institute":[],"abstract":"In this paper we propose a model, called lemonDIA, for representing lexical semantic change using the lemon framework and based on the ontological notion of the perdurant. Namely we extend the notion of sense in lemon by adding a temporal dimension and then define a class of perdurant entities that represents a shift in meaning of a word and which contains different related senses. We start by discussing the general problem of semantic shift and the utility of being able to easily access and represent such information in diachronic lexical resources. We then describe our model and illustrate it with examples","keywords":[],"pages":"","url":"https:\/\/iris.cnr.it\/handle\/20.500.14243\/259370","volume":"","doi":"","editors":[],"editors_source":"","published":"","publisher":"","issn":"","isbn":"","conference_name":"","conference_place":"","conference_date":"","last_updated_cnr":"2024-05-07 11:14:54","last_updated_oai":"0000-00-00 00:00:00","last_updated_www":"0000-00-00 00:00:00"},{"id":2376,"id_source":222787,"institutes":["ILC"],"type":"conference_article","type_order":7,"title":"The IMAGACT Visual Ontology. an Extendable Multilingual Infrastructure for the Representation of Lexical Encoding of Action","year":2014,"authors":["Moneglia, M.","Brown, S.","Frontini, F.","Gagliardi, G.","Khan, F.","Monachini, M.","Panunzi, A."],"authors_source":"Massimo Moneglia; Susan Brown; Francesca Frontini; Gloria Gagliardi; Fahad Khan; Monica Monachini;Alessandro Panunzi","authors_cnr_name":["FRONTINI, FRANCESCA","MONACHINI, MONICA"],"authors_cnr_id":["rp02790","rp19457"],"authors_cnr_institute":[],"abstract":"Action verbs have many meanings, covering actions in different ontological types. Moreover, each language categorizes action in its own way. One verb can refer to many different actions and one action can be identified by more than one verb. The range of variations within and across languages is largely unknown, causing trouble for natural language processing tasks. IMAGACT is a corpus-based ontology of action concepts, derived from English and Italian spontaneous speech corpora, which makes use of the universal language of images to identify the different action types extended by verbs referring to action in English, Italian, Chinese and Spanish. This paper presents the infrastructure and the various linguistic information the user can derive from it. IMAGACT makes explicit the variation of meaning of action verbs within one language and allows comparisons of verb variations within and across languages. Because the action concepts are represented with videos, extension into new languages beyond those presently implemented in IMAGACT is done using competence-based judgments by mother-tongue informants without intense lexicographic work involving underdetermined semantic description","keywords":[],"pages":"","url":"https:\/\/iris.cnr.it\/handle\/20.500.14243\/222787","volume":"","doi":"","editors":[],"editors_source":"","published":"","publisher":"","issn":"","isbn":"978-2-9517408-8-4","conference_name":"","conference_place":"","conference_date":"","last_updated_cnr":"2024-05-29 09:35:36","last_updated_oai":"0000-00-00 00:00:00","last_updated_www":"0000-00-00 00:00:00"},{"id":1542,"id_source":222825,"institutes":["ILC"],"type":"conference_article","type_order":7,"title":"Presenting a System of Human-Machine Interaction for Performing Map Tasks","year":2014,"authors":["Pallotti, G.","Frontini, F.","Aff\u00e8, F.","Monachini, M.","Ferrari, S."],"authors_source":"Pallotti, Gabriele; Frontini, Francesca; Aff\u00e8, Fabio; Monachini, Monica; Ferrari, Stefania","authors_cnr_name":["FRONTINI, FRANCESCA","MONACHINI, MONICA"],"authors_cnr_id":["rp02790","rp19457"],"authors_cnr_institute":[],"abstract":"A system for human machine interaction is presented, that offers second language learners of Italian the possibility of assessing their competence by performing a map task, namely by guiding the a virtual follower through a map with written instructions in natural language. The underlying natural language processing algorithm is described, and the map authoring infrastructure is presented","keywords":["Language learning","human machine interaction","map tasks"],"pages":"3963-3966","url":"http:\/\/www.lrec-conf.org\/proceedings\/lrec2014\/index.html","volume":"","doi":"","editors":["Calzolari, N.","Choukri, K.","Declerck, T.","Loftsson, H.","Maegaard, B.","Mariani, J.","Moreno, A.","Odijk, J.","Piperidis, S."],"editors_source":"N. Calzolari, K. Choukri, T. Declerck, H. Loftsson, B. Maegaard, J. Mariani, A. Moreno, J. Odijk, S. Piperidis","published":"","publisher":"European Language Resources Association ELRA (Paris, FRA)","issn":"","isbn":"978-2-9517408-8-4","conference_name":"9th International Conference on Language Resources and Evaluation, LREC 2014","conference_place":"Paris","conference_date":"","last_updated_cnr":"2024-05-31 11:42:51","last_updated_oai":"2024-05-31 11:42:51","last_updated_www":"0000-00-00 00:00:00"},{"id":1298,"id_source":265502,"institutes":["ILC"],"type":"conference_misc","type_order":8,"title":"Marine Planning and Service Platform (MAPS): An Advanced Research Engine for Grey Literature in Marine Science","year":2014,"authors":["Goggi, S.","Monachini, M.","Frontini, F.","Bartolini, R.","Pardelli, G.","De Mattei, M.","Bustaffa, F.","Manzella, G."],"authors_source":"Goggi, S; Monachini, M; Frontini, F; Bartolini, R; Pardelli, G; De Mattei, M; Bustaffa, F; Manzella, G","authors_cnr_name":["GOGGI, SARA","MONACHINI, MONICA","FRONTINI, FRANCESCA","BARTOLINI, ROBERTO","PARDELLI, GABRIELLA"],"authors_cnr_id":["rp20855","rp19457","rp02790","rp00239","rp20820"],"authors_cnr_institute":[],"abstract":"The MAPS (Marine Planning and Service Platform) project is a development of the Marine project (Ricerca Industriale e Sviluppo Sperimentale Regione Liguria 2007-2013) aiming at building a computer platform for supporting Operative Oceanography in its activities. One of the main objective of the project is to develop a repository that should gather, classify and structure marine scientific literature and data thus guaranteeing their accessibility to researchers and institutions by means of standard protocols. Community and Requirements. Operative Oceanography is the branch of marine research which deals with the development of integrated systems for examining and modeling the ocean monitoring and forecast. Experts need access to real-time data on the state of the sea such as forecasts on temperatures, streams, tides and the relevant scientific literature. This finds application in many areas, ranging from civilian and military safety to protection of off-shore and coastal infrastructures. The metadata. The set of metadata associated with marine data is defined in the CDI (Common Data Index) documented standard. They encode: the types of sizes which have been measured; the measurement tools the platform which has been employed; the geographic area where measures have been taken; the environmental matrix; the descriptive documentation. As concerns the scientific documentation, at the current stage of the CDI standard, a document is shaped around the following metadata: Title, Authors, Version, ISBN\/DOI, Topic, Date of publication, Body\/Institution, Abstract. The search engine. The query system (which is actually under development) has been designed for operating with structured data-the metadata-and raw data-the associated technical and scientific documentation. Full-text technologies are often unsuccessful when applied to this type of queries since they assume the presence of specific keywords in the text; in order to fix this problem, the MAPS project suggests to use different emantic technologies for retrieving the text and data and thus getting much more complying results. In the Poster we will present the scenario of the Operative Oceanography together with the technologies used to develop an advanced earch engine which aims at providing rapid and efficient access to a Digital Library of oceanographic data. The case-study is also highlighting how the retrieval of grey literature from this specific marine community could be reproduced for similar communities as well, thus revealing the 2 great impact that the processing, re-use as well as application of grey data have on societal needs\/problems and their answers","keywords":["Marine Science","Search Engine","Source Data","Oceanography"],"pages":"93-94","url":"http:\/\/greyguide.isti.cnr.it\/dfdownloadnew.php?ident=GLConference\/GL16\/2014-G01-015&langver=en&scelta=Metadata","volume":"","doi":"","editors":["Farace, C. B. D.","Frantzen, J."],"editors_source":"compiled by D. Farace and J. Frantzen","published":"","publisher":"","issn":"","isbn":"978-90-77484-24-1","conference_name":"Sixteenth International Conference on Grey Literature Grey Literature Lobby: Engines and Requesters for Change","conference_place":"","conference_date":"","last_updated_cnr":"2024-06-02 11:12:43","last_updated_oai":"2024-06-02 11:12:43","last_updated_www":"0000-00-00 00:00:00"},{"id":1655,"id_source":276258,"institutes":["ILC"],"type":"technical_report","type_order":9,"title":"MAPS: Architettura del Sistema","year":2014,"authors":["De Mattei, M.","Medone, D.","D'Angelo, P.","Monachini, M.","Bartolini, R.","Frontini, F."],"authors_source":"M. De Mattei; D. Medone; P. D'Angelo; M. Monachini; R. Bartolini; F. Frontini","authors_cnr_name":["MONACHINI, MONICA","BARTOLINI, ROBERTO","FRONTINI, FRANCESCA"],"authors_cnr_id":["rp19457","rp00239","rp02790"],"authors_cnr_institute":[],"abstract":"PROGRAMMA OPERATIVO REGIONALE POR-FESR (2007-2013) Asse 1 Innovazione e Competitivit\u00e0 Bando DLTM Azione 1. 2. 2 \"Ricerca industriale e sviluppo sperimentale a favore delle imprese del Distretto Ligure per le Tecnologie Marine (DLTM) anno 2012. Il presente documento \u00e8 il deliverable \"D3. 1-Architettura del Sistema\" del progetto MAPS (Marine Planning and Service Platform). Il progetto MAPS \u00e8 un'evoluzione del progetto precedente Marine. Tale evoluzione si articola su tre aspetti diversi:-Un meccanismo di federazione dei dati, che consenta di rendere disponibili ai propri utenti non soltanto i dati prodotti internamente da sistema Marine ma anche quelli resi disponibili da altri sistemi similari, soddisfacendo cos\u00ec un pi\u00f9 ampio ambito di esigenze informative. Il deliverable D2. 2, Modello della Soluzione specifica in dettaglio queste nuove funzionalit\u00e0.-Un Catalogo dei Documenti che, conservando la documentazione tecnica e scientifica dei prodotti offerti, possa documentare in modo accurato le modalit\u00e0 di misurazione, elaborazione e controllo dei prodotti forniti e quindi i relativi ambiti di applicabilit\u00e0.-Un sistema di ricerca capace di selezionare i dati necessari ad uno scopo determinato non soltanto sulla base della loro tipologia, della loro dislocazione territoriale o di altre informazioni simili contenute nei metadati associati come avviene oggi nella maggior parte dei sistemi esistenti, ma anche sulla base delle informazioni contenute nella documentazione tecnica e scientifica. Tali funzionalit\u00e0 sono specificate nel deliverable D1. 3-Modello della Soluzione","keywords":["Marine Science","Search Engine","Source Data","Oceanography"],"pages":"1-35","url":"https:\/\/iris.cnr.it\/handle\/20.500.14243\/276258","volume":"","doi":"","editors":[],"editors_source":"","published":"","publisher":"","issn":"","isbn":"","conference_name":"","conference_place":"","conference_date":"","last_updated_cnr":"2024-05-01 11:21:00","last_updated_oai":"2024-05-01 11:21:00","last_updated_www":"0000-00-00 00:00:00"},{"id":440,"id_source":276262,"institutes":["ILC"],"type":"technical_report","type_order":9,"title":"META: Report di progettazione degli algoritmi individuati","year":2014,"authors":["De Mattei, M.","Medone, D.","Maltese, M.","Frontini, F.","Bartolini, R.","Monachini, M."],"authors_source":"De Mattei, Maurizio; Medone, Daniele; Maltese, Maurizio; Frontini, Francesca; Bartolini, Roberto; Monachini, Monica","authors_cnr_name":["FRONTINI, FRANCESCA","BARTOLINI, ROBERTO","MONACHINI, MONICA"],"authors_cnr_id":["rp02790","rp00239","rp19457"],"authors_cnr_institute":[],"abstract":"PROGRAMMA OPERATIVO REGIONALE POR-FESR (2007-2013) Asse 1 Innovazione e Competitivit\u00e0 Bando DLTM Azione 1. 2. 2 \"Ricerca industriale e sviluppo sperimentale a favore delle imprese del Distretto Ligure per le Tecnologie Marine (DLTM) anno 2012. Il deliverable definisce l'architettura del Sistema di Estrazione Eventi Meteo realizzato dagli autori nell'ambito del progetto META. Il sistema estrae da contenuti online informazione su eventi meteo critici verificatesi in Liguria e nel nord della Toscana","keywords":["Ontology","Information Extraction","Taxonomy"],"pages":"1-19","url":"https:\/\/iris.cnr.it\/handle\/20.500.14243\/276262","volume":"","doi":"","editors":[],"editors_source":"","published":"","publisher":"","issn":"","isbn":"","conference_name":"","conference_place":"","conference_date":"","last_updated_cnr":"2024-04-06 11:34:18","last_updated_oai":"2024-04-06 11:34:18","last_updated_www":"0000-00-00 00:00:00"},{"id":760,"id_source":276261,"institutes":["ILC"],"type":"technical_report","type_order":9,"title":"META:-Report sui modelli e tecniche linguistiche","year":2014,"authors":["Frontini, F.","Bartolini, R.","Monachini, M."],"authors_source":"Frontini, Francesca; Bartolini, Roberto; Monachini, Monica","authors_cnr_name":["FRONTINI, FRANCESCA","BARTOLINI, ROBERTO","MONACHINI, MONICA"],"authors_cnr_id":["rp02790","rp00239","rp19457"],"authors_cnr_institute":[],"abstract":"PROGRAMMA OPERATIVO REGIONALE POR-FESR (2007-2013) Asse 1 Innovazione e Competitivit\u00e0 Bando DLTM Azione 1. 2. 2 \"Ricerca industriale e sviluppo sperimentale a favore delle imprese del Distretto Ligure per le Tecnologie Marine (DLTM) anno 2012. Il deliverable riassume lo stato dell'arte delle tecnologie semantiche che possono essere impiegate nella realizzazione del progetto META. Il progetto META \u00e8 una progetto di ricerca e sviluppo tecnologico finanziato dalla Regione Liguria con i fondi POR-FESR 2007-2013 della Comunit\u00e0 Europea che mira alla realizzazione di un sistema per l'allerta di eventi meteo critici in Liguria e nel nord della Toscana. Nell'ambito del progetto META le tecnologie semantiche sono utilizzate per estrarre eventi meteo di interesse da articoli pubblicati in rete o sui social network","keywords":["Ontology","Information Extraction","Semantic Web","Search Engine"],"pages":"1-20","url":"https:\/\/iris.cnr.it\/handle\/20.500.14243\/276261","volume":"","doi":"","editors":[],"editors_source":"","published":"","publisher":"","issn":"","isbn":"","conference_name":"","conference_place":"","conference_date":"","last_updated_cnr":"2024-05-29 09:11:46","last_updated_oai":"2024-05-29 09:11:46","last_updated_www":"0000-00-00 00:00:00"},{"id":2074,"id_source":276259,"institutes":["ILC"],"type":"technical_report","type_order":9,"title":"MAPS: Stato dell'Arte","year":2014,"authors":["Frontini, F.","Bartolini, R.","Monachini, M."],"authors_source":"Frontini, Francesca; Bartolini, Roberto; Monachini, Monica","authors_cnr_name":["FRONTINI, FRANCESCA","BARTOLINI, ROBERTO","MONACHINI, MONICA"],"authors_cnr_id":["rp02790","rp00239","rp19457"],"authors_cnr_institute":[],"abstract":"PROGRAMMA OPERATIVO REGIONALE POR-FESR (2007-2013) Asse 1 Innovazione e Competitivit\u00e0 Bando DLTM Azione 1. 2. 2 \"Ricerca industriale e sviluppo sperimentale a favore delle imprese del Distretto Ligure per le Tecnologie Marine (DLTM) anno 2012 Il documento descrive lo stato dell'arte delle tecnologie linguistiche applicate ai sistemi di ricerca semantica","keywords":["Marine Science","Search Engine","Source Data","Oceanography"],"pages":"1-21","url":"https:\/\/iris.cnr.it\/handle\/20.500.14243\/276259","volume":"","doi":"","editors":[],"editors_source":"","published":"","publisher":"","issn":"","isbn":"","conference_name":"","conference_place":"","conference_date":"","last_updated_cnr":"2024-05-01 11:27:23","last_updated_oai":"2024-05-01 11:27:23","last_updated_www":"0000-00-00 00:00:00"},{"id":938,"id_source":222835,"institutes":["ILC"],"type":"technical_report","type_order":9,"title":"Stato dell'arte dei motori semantici. Progetto MAPS, programma operativo regionale POR-FESR (2007-2013)","year":2014,"authors":["Frontini, F.","Bartolini, R.","Monachini, M.","Pardelli, G.","Goggi, S."],"authors_source":"Francesca Frontini; Roberto Bartolini; Monica Monachini; Gabriella Pardelli; Sara Goggi","authors_cnr_name":["FRONTINI, FRANCESCA","BARTOLINI, ROBERTO","MONACHINI, MONICA","PARDELLI, GABRIELLA","GOGGI, SARA"],"authors_cnr_id":["rp02790","rp00239","rp19457","rp20820","rp20855"],"authors_cnr_institute":[],"abstract":"Il presente documento \u00e8 il deliverable \"D1. 1-Stato dell'Arte dei motori semantici del progetto MAPS (Marine Planning and Service Platform). Il progetto MAPS \u00e8 una evoluzione del progetto precedente Marine. Tramite il progetto Marine (Bando Ricerca Industriale e Sviluppo Sperimentale Regione Liguria 2007-2013-pos n. 1) \u00e8 stata realizzata una piattaforma informatica di supporto all'Oceanografia Operativa capace di raccogliere dati marini per renderli poi disponibili ai ricercatori e alle organizzazioni interessate tramite protocolli standard. Lo scopo del progetto MAPS \u00e8 quello di realizzare una Catalogo di Documenti contenente informazioni per la piattaforma Marine. Caratteristica di MAPS \u00e8 di fornire accesso ai dati oceanografici sia attraverso la ricerca per metadati, sia attraverso la ricerca semantica contenuta nella manualistica tecnico scientifica di riferimento","keywords":[],"pages":"1-22","url":"https:\/\/iris.cnr.it\/handle\/20.500.14243\/222835","volume":"","doi":"","editors":[],"editors_source":"","published":"","publisher":"","issn":"","isbn":"","conference_name":"","conference_place":"","conference_date":"","last_updated_cnr":"2024-05-28 15:31:05","last_updated_oai":"2024-05-28 15:31:05","last_updated_www":"0000-00-00 00:00:00"},{"id":577,"id_source":286128,"institutes":["ILC"],"type":"misc","type_order":11,"title":"La mappa delle opinioni e dei sentimenti estratte dai social media","year":2014,"authors":["Frontini, F."],"authors_source":"Frontini, Francesca","authors_cnr_name":["FRONTINI, FRANCESCA"],"authors_cnr_id":["rp02790"],"authors_cnr_institute":[],"abstract":"","keywords":[],"pages":"","url":"https:\/\/iris.cnr.it\/handle\/20.500.14243\/286128","volume":"","doi":"","editors":[],"editors_source":"","published":"","publisher":"","issn":"","isbn":"","conference_name":"Seminario rivolto agli alunni dell'Istituto Tecnico Economico \"F. Carrara\" di Lucca, organizzato dall'Istituto di Linguistica Computazionale \"A. Zampolli\" del CNR di Pisa","conference_place":"","conference_date":"","last_updated_cnr":"2024-05-12 11:36:30","last_updated_oai":"2024-05-12 11:36:30","last_updated_www":"0000-00-00 00:00:00"},{"id":1497,"id_source":262584,"institutes":["ILC"],"type":"misc","type_order":11,"title":"A Model for Representing Diachronic Semantic Information in Lexico-Semantic Resources on the Semantic Web","year":2014,"authors":["Khan, F.","Frontini, F.","Monachini, M."],"authors_source":"Khan F.; Frontini F.; Monachini M.","authors_cnr_name":["KHAN, ANAS FAHAD ASLAM","FRONTINI, FRANCESCA","MONACHINI, MONICA"],"authors_cnr_id":["rp05508","rp02790","rp19457"],"authors_cnr_institute":[],"abstract":"The Semantic Web offers a way of publishing structured data online that facilitates the interlinking of different datasets stored at different online locations? indeed one of the main aims of the Semantic Web movement is to actively encourage this enrichment of online datasets with information from other resources, in order to avoid the problem of so called 'data islands'. In contrast to conventional hyperlinks however the links between different resources on the Semantic Web can be given semantic types and classified hierarchically. Data published on the Semantic Web is referred to as Linked Data? if, in addition, this data is available with an open license then it can be referred to as Linked Open Data (Heath 2011)","keywords":["Cultural resources","Heritage resources"],"pages":"1-3","url":"http:\/\/www.dh.uni-leipzig.de\/wo\/wp-content\/uploads\/2014\/11\/Fahad-Khan-Francesca-Frontini-and-Monica-Monachini-A-Model-for-Representing.pdf","volume":"","doi":"","editors":[],"editors_source":"","published":"","publisher":"","issn":"","isbn":"","conference_name":"Greek and Latin in an age of Open Data. Open Philology Project","conference_place":"","conference_date":"","last_updated_cnr":"2025-01-24 23:17:43","last_updated_oai":"2025-01-24 23:17:43","last_updated_www":"0000-00-00 00:00:00"},{"id":1488,"id_source":226376,"institutes":["ILC"],"type":"conference_article","type_order":7,"title":"Linking the Geonames ontology to WordNet","year":2013,"authors":["Frontini, F.","Del Gratta, R.","Monachini, M."],"authors_source":"Francesca Frontini; Riccardo Del Gratta; Monica Monachini.","authors_cnr_name":["FRONTINI, FRANCESCA","DEL GRATTA, RICCARDO","MONACHINI, MONICA"],"authors_cnr_id":["rp02790","rp00284","rp19457"],"authors_cnr_institute":[],"abstract":"This paper illustrates the transformation of the GeoNames ontology concepts, with their English labels and glosses, into a GeoDomain WordNet-like resource in English, its translation into Italian, and its linking to the existing generic WordNets of both languages","keywords":["GeoNames","WordNet","lemon"],"pages":"263-267","url":"http:\/\/hnk.ffzg.hr\/bibl\/ltc2013\/book\/papers\/OWN-2.pdf","volume":"","doi":"","editors":["Vetulani, Z.","Uszkoreit, H."],"editors_source":"Zygmunt Vetulani & Hans Uszkoreit (ed.)","published":"Human Language Technologies as a Challenge for Computer Science and Linguistics. Proceedings, 6th Language & Technology Conference, December 7-9, 2013, Poznan\u0303, Poland","publisher":"Fundacja Uniwersytetu im A. Mickiewicza (Poznan, POL)","issn":"","isbn":"978-2-9517408-8-4","conference_name":"6th Language & Technology Conference: Human Language Technologies as a Challenge for Computer Science and Linguistics","conference_place":"Poznan","conference_date":"","last_updated_cnr":"2024-10-02 12:43:50","last_updated_oai":"2024-10-02 12:43:50","last_updated_www":"0000-00-00 00:00:00"},{"id":2416,"id_source":259365,"institutes":["ILC"],"type":"conference_article","type_order":7,"title":"Generative Lexicon Theory and Linguistic Linked Open Data","year":2013,"authors":["Khan, F.","Frontini, F.","Del Gratta, R.","Monachini, M.","Quochi, V."],"authors_source":"Fahad Khan; Francesca Frontini; Riccardo Del Gratta; Monica Monachini; Valeria Quochi","authors_cnr_name":["FRONTINI, FRANCESCA","DEL GRATTA, RICCARDO","MONACHINI, MONICA","QUOCHI, VALERIA"],"authors_cnr_id":["rp02790","rp00284","rp19457","rp13283"],"authors_cnr_institute":[],"abstract":"In this paper we look at how Generative Lexicon theory can assist in providing a more thorough definition of word senses as links between items in a RDF-based lexicon and concepts in an ontology. We focus on the definition of lexical sense in lemon and show its limitations before defining a new model based on lemon and which we term lemonGL. This new model is an initial attempt at providing a way of structuring lexico-ontological resources as linked data in such a way as to allow a rich representation of word meaning (following the GL theory) while at the same time (attempting to) re-main faithful to the separation between the lexicon and the ontology as recommended by the lemon model","keywords":[],"pages":"","url":"https:\/\/iris.cnr.it\/handle\/20.500.14243\/259365","volume":"","doi":"","editors":[],"editors_source":"","published":"","publisher":"","issn":"","isbn":"978-1-937284-98-5","conference_name":"","conference_place":"","conference_date":"","last_updated_cnr":"2024-10-02 12:46:03","last_updated_oai":"0000-00-00 00:00:00","last_updated_www":"0000-00-00 00:00:00"},{"id":126,"id_source":226423,"institutes":["IIT","ILC"],"type":"conference_article","type_order":7,"title":"Tour-pedia: a web application for the analysis and visualization of opinions for tourism domain","year":2013,"authors":["Marchetti, A.","Tesconi, M.","Abbate, S.","Lo Duca, A.","D'Errico, A.","Frontini, F.","Monachini, M."],"authors_source":"Marchetti, Andrea; Tesconi, Maurizio; Abbate, Stefano; Lo Duca, Angelica; D'Errico, Andrea; Frontini, Francesca; Monachini, Monica","authors_cnr_name":["MARCHETTI, ANDREA","TESCONI, MAURIZIO","LO DUCA, ANGELICA","FRONTINI, FRANCESCA","MONACHINI, MONICA"],"authors_cnr_id":["rp24107","rp04403","rp04761","rp02790","rp19457"],"authors_cnr_institute":[],"abstract":"We present Tour-pedia an interactive web application that extracts opinions from reviews of accommodations from different sources available on-line. Polarity markers display on a map the different opinions. This tool is intended to help business operators to manage reputation on-line","keywords":["Visualization tools","opinion mining","NLP on social media","tourism reviews"],"pages":"594-595","url":"http:\/\/www.iit.cnr.it\/sites\/default\/files\/ltc2013_opener_demo.pdf","volume":"","doi":"","editors":["Vetulani, Z.","Uszkoreit, H."],"editors_source":"Zygmunt Vetulani & Hans Uszkoreit (ed.)","published":"","publisher":"Fundacja Uniwersytetu im A. Mickiewicza (Poznan, POL)","issn":"","isbn":"978-83-932640-4-9","conference_name":"6th Language & Technology Conference: Human Language Technologies as a Challenge for Computer Science and Linguistics","conference_place":"Poznan","conference_date":"","last_updated_cnr":"2025-09-27 00:55:02","last_updated_oai":"2025-09-27 00:55:02","last_updated_www":"0000-00-00 00:00:00"},{"id":1999,"id_source":226438,"institutes":["ILC"],"type":"conference_article","type_order":7,"title":"IMAGACT E-learning Platform for Basic Action Types. In: Pixel (ed.), Proceedings of the 6th International Conference ICT for Language Learning","year":2013,"authors":["Moneglia, M.","Panunzi, A.","Gagliardi, G.","Monachini, M.","Russo, I.","De Felice, I.","Khan, F.","Frontini, F."],"authors_source":"Moneglia M.; Panunzi A.; Gagliardi G.; Monachini M.; Russo I.; De Felice I.; Khan F.;Frontini F.","authors_cnr_name":["MONACHINI, MONICA","RUSSO, IRENE","DE FELICE, IRENE","KHAN, ANAS FAHAD ASLAM","FRONTINI, FRANCESCA"],"authors_cnr_id":["rp19457","rp02389","rp05254","rp05508","rp02790"],"authors_cnr_institute":[],"abstract":"Action verbs express important information in a sentence and they are the most frequent elements in speech, but they are also one of the most difficult part of the lexicon to learn for L2 language learners, because languages segment these concepts in very different ways. The two sentences \"Mary folds her shirt\" and \"Mary folds her arms\" refer to two completely different types of action, as becomes evident when they are translated into another language (e. g., in Italian they would be translated as \"Maria piega la camicia\" and \"Maria incrocia le braccia\" respectively). IMAGACT e-learning platform aims to make these differences evident by creating a cross-linguistic ontology of action types, whose nodes consist of 3D scenes, each of which relates to one action type. In order to identify these types, contexts of use have been extracted from English and Italian spontaneous speech corpora for around 600 high frequency action verbs (for each language). All instances that refer to similar events (e. g., fold the shirt\/ the blanket) are grouped under one single action type: each one of these types is then represented by a linguistic best example and a short video that represents simple actions (e. g. a man taking a glass from a table). The action types extracted for Italian and English are compared and merged into one cross-linguistic ontology of action. IMAGACT has provided an internet based annotation infrastructure to derive this information from corpora. The project is now completed for the Italian and English lexicon, data extraction for Chinese and Spanish is ongoing. Reference to prototypical imagery is crucial in order to bootstrap the learning process. By selecting the set of 3D scenes referred to by a verb in one language and viewing the type of activity represented therein learners can directly understand the range of applicability of each verb. Thanks to an easy interface, a user can access the English\/Italian\/Chinese lexicon by lemma or directly by 3D scenes. For example, searching for the verb \"to turn\", s\/he will be presented with a number of scenes, showing the various action types associated to that verb. Clicking on a scene s\/he or she will know how this type of action is referred to in other the languages","keywords":["Ontology"],"pages":"85-89","url":"https:\/\/iris.cnr.it\/handle\/20.500.14243\/226438","volume":"","doi":"","editors":["Pixel"],"editors_source":"Pixel (ed.)","published":"Conference Proceedings. ICT for Language Learning","publisher":"libreriauniversitaria. it (Limena, ITA)","issn":"","isbn":"978-88-6292-423-8","conference_name":"International Conference \"ICT for Language Learning\", 6th edition","conference_place":"Limena","conference_date":"","last_updated_cnr":"2025-01-24 02:16:21","last_updated_oai":"2025-01-24 02:16:21","last_updated_www":"0000-00-00 00:00:00"},{"id":299,"id_source":257360,"institutes":["ILC"],"type":"conference_article","type_order":7,"title":"(Fore)seeing actions in objects. Acquiring distinctive affordances from language","year":2013,"authors":["Russo, I.","De Felice, I.","Frontini, F.","Khan, F.","Monachini, M."],"authors_source":"Russo, Irene; DE FELICE, Irene; Frontini, Francesca; Khan, Fahad; Monachini, Monica","authors_cnr_name":["RUSSO, IRENE","DE FELICE, IRENE","FRONTINI, FRANCESCA","MONACHINI, MONICA"],"authors_cnr_id":["rp02389","rp05254","rp02790","rp19457"],"authors_cnr_institute":[],"abstract":"In this paper we investigate if conceptual information concerning objects' affordances as possibilities for actions anchored to an object can be at least partially acquired through language. Considering verb-noun pairs as the linguistic realizations of relations between actions performed by an agent and objects we collect this information from the ImagAct dataset, a linguistic resource obtained from manual annotation of basic action verbs, and from a web corpus(itTenTen). The notion of affordance verb as the most distinctive verb in ImagAct enables a comparison with distributional data that reveal how lemmas ranking based on a semantic association measure that mirror that of affordances as the most distinctive actions an object can be involved in","keywords":[],"pages":"151-161","url":"https:\/\/docs.google.com\/viewer?a=v&pid=sites&srcid=ZGVmYXVsdGRvbWFpbnxubHBjczIwMTN8Z3g6MTI0ZGMzYWYwYmMxNjY1Mg","volume":"","doi":"","editors":["Sharp, B.","Zock, M."],"editors_source":"Bernadette Sharp, Michael Zock","published":"Proceedings of NLPCS 2013-10th International Workshop on Natural Language Processing and Cognitive Science","publisher":"","issn":"","isbn":"","conference_name":"NLPCS 2013-10th International Workshop on Natural Language Processing and Cognitive Science","conference_place":"","conference_date":"","last_updated_cnr":"2024-05-15 14:04:24","last_updated_oai":"2024-05-15 14:04:24","last_updated_www":"0000-00-00 00:00:00"},{"id":410,"id_source":227078,"institutes":["ILC"],"type":"conference_article","type_order":7,"title":"Disambiguation of Basic Action Types through Nouns' Telic Qualia","year":2013,"authors":["Russo, I.","Frontini, F.","De Felice, I.","Khan, F.","Monachini, M."],"authors_source":"Russo, Irene; Frontini, Francesca; DE FELICE, Irene; Khan, Fahad; Monachini, Monica","authors_cnr_name":["RUSSO, IRENE","FRONTINI, FRANCESCA","DE FELICE, IRENE","MONACHINI, MONICA"],"authors_cnr_id":["rp02389","rp02790","rp05254","rp19457"],"authors_cnr_institute":[],"abstract":"Knowledge about semantic associations between words is effective to disambiguate word senses. The aim of this paper is to investigate the role and the relevance of telic information from SIMPLE in the disambiguation of basic action types of Italian HOLD verbs (prendere, 'to take', raccogliere, 'to pick up', pigliare 'to grab' etc.). We propose an experiment to compare the results obtained with telic information from SIMPLE with basic co-occurrence information extracted from corpora (most salient verbs modifying nouns) classified in terms of general semantic classes to avoid data sparseness","keywords":[],"pages":"70-75","url":"http:\/\/www.aclweb.org\/anthology\/W13-5410","volume":"","doi":"","editors":["Sauri\u0301, R.","Calzolari, N.","Huang, C. R.","Lenci, A.","Monachini, M.","Pustejovsky, J."],"editors_source":"Roser Sauri\u0301, Nicoletta Calzolari, Chu-Ren Huang, Alessandro Lenci, Monica Monachini, James Pustejovsky","published":"Proceedings of the 6th International Conference on Generative Approaches to the Lexicon. Generative Lexicon and Distributional Semantics","publisher":"Association for Computational Linguistics (Stroudsburg, USA)","issn":"","isbn":"978-1-937284-98-5","conference_name":"6th International Conference on Generative Approaches to the Lexicon Generative Lexicon and Distributional Semantics","conference_place":"Stroudsburg","conference_date":"","last_updated_cnr":"2024-05-19 13:48:04","last_updated_oai":"2024-05-19 13:48:04","last_updated_www":"0000-00-00 00:00:00"},{"id":1668,"id_source":222834,"institutes":["ILC"],"type":"conference_article","type_order":7,"title":"Flexible Acquisition of Subcategorization Frames in Italian","year":2012,"authors":["Caselli, T.","Frontini, F.","Quochi, V.","Rubino, F.","Russo, I."],"authors_source":"Caselli, Tommaso; Frontini, Francesca; Quochi, Valeria; Rubino, Francesco; Russo, Irene","authors_cnr_name":["FRONTINI, FRANCESCA","QUOCHI, VALERIA","RUSSO, IRENE"],"authors_cnr_id":["rp02790","rp13283","rp02389"],"authors_cnr_institute":[],"abstract":"Lexica of predicate-argument structures constitute a useful tool for several tasks in NLP. This paper describes a web-service system for automatic acquisition of verb subcategorization frames (SCFs) from parsed data in Italian. The system acquires SCFs in an unsupervised manner. We created two gold standards for the evaluation of the system, the first by mixing together information from two lexica (one manually created and the second automatically acquired) and manual exploration of corpus data and the other annotating data extracted from a specialized corpus (environmental domain). Data filtering is accomplished by means of the maximum likelihood estimate (MLE). The evaluation phase has allowed us to identify the best empirical MLE threshold for the creation of a lexicon (P=0. 653, R=0. 557, F1=0. 601). In addition to this, we assigned to the extracted entries of the lexicon a confidence score based on the relative frequency and evaluated the extractor on domain specific data. The confidence score will allow the final user to easily select the entries of the lexicon in terms of their reliability: one of the most interesting feature of this work is the possibility the final users have to customize the results of the SCF extractor, obtaining different SCF lexica in terms of size and accuracy","keywords":["lexicon","automatic acquisition","subcategorisation frames"],"pages":"2842-2848","url":"http:\/\/www.lrec-conf.org\/proceedings\/lrec2012\/summaries\/390.html","volume":"","doi":"","editors":["Calzolari, N.","Choukri, K.","Declerck, T.","Do\u011fan, M. U.","Maegaard, B.","Mariani, J.","Odijk, J.","Piperidis, S."],"editors_source":"Nicoletta Calzolari, Khalid Choukri, Thierry Declerck, Mehmet U?ur Do?an, Bente Maegaard, Joseph Mariani, Jan Odijk, Stelios Piperidis","published":"Proceedings of the Eight International Conference on Language Resources and Evaluation (LREC'12)","publisher":"European Language Resources Association ELRA (Paris, FRA)","issn":"","isbn":"9782951740877","conference_name":"Eight International Conference on Language Resources and Evaluation (LREC'12)","conference_place":"Paris","conference_date":"","last_updated_cnr":"2025-03-07 22:38:43","last_updated_oai":"2025-03-07 22:38:43","last_updated_www":"0000-00-00 00:00:00"},{"id":69,"id_source":117790,"institutes":["ILC","IIT"],"type":"conference_article","type_order":7,"title":"L-LEME: an Automatic Lexical Merger based on the LMF Standard","year":2012,"authors":["Del Gratta, R.","Frontini, F.","Monachini, M.","Quochi, V.","Rubino, F.","Abrate, M.","Lo Duca, A."],"authors_source":"DEL GRATTA, Riccardo; Frontini, Francesca; Monachini, Monica; Quochi, Valeria; Rubino, Francesco; Abrate, Matteo; LO DUCA, Angelica","authors_cnr_name":["DEL GRATTA, RICCARDO","FRONTINI, FRANCESCA","MONACHINI, MONICA","QUOCHI, VALERIA","ABRATE, MATTEO","LO DUCA, ANGELICA"],"authors_cnr_id":["rp00284","rp02790","rp19457","rp13283","rp02900","rp04761"],"authors_cnr_institute":[],"abstract":"The present paper describes LMF LExical MErger (L-LEME), an architecture to combine two lexicons in order to obtain new resource(s). L-LEME relies on standards, thus exploiting the benefits of the ISO Lexical Markup Framework (LMF) to ensure interoperability. L-LEME is meant to be dynamic and heavily adaptable: it allows the users to configure it to meet their specific needs. The L-LEME architecture is composed of two main modules: the Mapper, which takes in input two lexicons A and B and a set of user-defined rules and instructions to guide the mapping process (Directives D) and gives in output all matching entries. The algorithm also calculates a cosine similarity score. The Builder takes in input the previous results, a set of Directives D1 and produces a new LMF lexicon C. The Directives allow the user to define its own building rules and different merging scenarios. L-LEME is applied to a specific concrete task within the PANACEA project, namely the merging of two Italian SubCategorization Frame (SCF) lexicons. The experiment is interesting in that A and B have different philosophies behind, being A built by human introspection and B automatically extracted. Ultimately, L-LEME has interesting repercussions in many language technology applications","keywords":["LMF","Lexicon mapping","similarity score"],"pages":"31-40","url":"https:\/\/iris.cnr.it\/handle\/20.500.14243\/117790","volume":"","doi":"","editors":["Bel, N.","Gavrilidou, M.","Monachini, M.","Quochi, V.","Rimell, L."],"editors_source":"Bel N. , Gavrilidou M. , Monachini M., Quochi V., Rimell L.","published":"Proceedings of the LREC 2012 Workshop on Language Resource Merging","publisher":"","issn":"","isbn":"978-2-9517408-7-7","conference_name":"The Eight International Conference on Language Resources and Evaluation (LREC) 2012","conference_place":"","conference_date":"","last_updated_cnr":"2024-10-02 13:15:11","last_updated_oai":"2024-10-02 13:15:11","last_updated_www":"0000-00-00 00:00:00"},{"id":1685,"id_source":119634,"institutes":["ILC"],"type":"conference_article","type_order":7,"title":"The Language Library: supporting community effort for collective resource production","year":2012,"authors":["Del Gratta, R.","Frontini, F.","Rubino, F.","Russo, I.","Calzolari, N."],"authors_source":"DEL GRATTA, Riccardo; Frontini, Francesca; Rubino, Francesco; Russo, Irene; Calzolari, Nicoletta","authors_cnr_name":["DEL GRATTA, RICCARDO","FRONTINI, FRANCESCA","RUSSO, IRENE"],"authors_cnr_id":["rp00284","rp02790","rp02389"],"authors_cnr_institute":[],"abstract":"Relations among phenomena at different linguistic levels are at the essence of language properties but today we focus mostly on one specific linguistic layer at a time, without (having the possibility of) paying attention to the relations among the different layers. At the same time our efforts are too much scattered without much possibility of exploiting other people's achievements. To address the complexities hidden in multilayer interrelations even small amounts of processed data can be useful, improving the performance of complex systems. Exploiting the current trend towards sharing we want to initiate a collective movement that works towards creating synergies and harmonisation among different annotation efforts that are now dispersed. In this paper we present the general architecture of the Language Library, an initiative which is conceived as a facility for gathering and making available through simple functionalities the linguistic knowledge the field is able to produce, putting in place new ways of collaboration within the LRT community. In order to reach this goal, a first population round of the Language Library has started around a core of parallel\/comparable texts that have been annotated by several contributors submitting a paper for LREC2012. The Language Library has also an ancillary aim related to language documentation and archiving and it is conceived as a theory-neutral space which allows for several language processing philosophies to coexist","keywords":["annotation","metadata","scientific crowdsourcing"],"pages":"43-49","url":"https:\/\/iris.cnr.it\/handle\/20.500.14243\/119634","volume":"","doi":"","editors":[],"editors_source":"","published":"The Eight International Conference on Language Resources and Evaluation (LREC'12)","publisher":"","issn":"","isbn":"","conference_name":"The Eight International Conference on Language Resources and Evaluation (LREC'12)","conference_place":"","conference_date":"","last_updated_cnr":"2025-03-07 22:50:08","last_updated_oai":"2025-03-07 22:50:08","last_updated_www":"0000-00-00 00:00:00"},{"id":747,"id_source":251924,"institutes":["ILC","IIT"],"type":"conference_article","type_order":7,"title":"GLOSS, an infrastructure for the semantic annotation and mining of documents in the public security domain","year":2012,"authors":["Frontini, F.","Aliprandi, C.","Bacciu, C.","Bartolini, R.","Marchetti, A.","Parenti, E.","Piccinonno, F.","Soru, T."],"authors_source":"Frontini Francesca; Aliprandi Carlo; Bacciu Clara; Bartolini Roberto; Marchetti Andrea; Parenti Enrico; Piccinonno Fulvio; Soru T.","authors_cnr_name":["FRONTINI, FRANCESCA","BACCIU, CLARA","BARTOLINI, ROBERTO","MARCHETTI, ANDREA"],"authors_cnr_id":["rp02790","rp02898","rp00239","rp24107"],"authors_cnr_institute":[],"abstract":"Efficient access to information is crucial in the work of organizations that require decision taking in emergency situations. This paper gives an outline of GLOSS, an integrated system for the analysis and retrieval of data in the environmental and public security domain. We shall briefly present the GLOSS infrastructure and its use, and how semantic information of various kinds is integrated, annotated and made available to the final users","keywords":["semantic annotation","text mining","geographic data"],"pages":"21-25","url":"https:\/\/iris.cnr.it\/handle\/20.500.14243\/251924","volume":"","doi":"","editors":[],"editors_source":"","published":"","publisher":"European language resources association (ELRA) (Paris, FRA)","issn":"","isbn":"978-2-9517408-7-7","conference_name":"Eight International Conference on Language Resources and Evaluation. LREC'12. European Language Resources Association: France","conference_place":"Paris","conference_date":"","last_updated_cnr":"2024-03-31 13:40:37","last_updated_oai":"2024-03-31 13:40:37","last_updated_www":"0000-00-00 00:00:00"},{"id":150,"id_source":128272,"institutes":["ILC"],"type":"conference_article","type_order":7,"title":"Automatic Creation of Quality Multi-Word Lexica from Noisy Text Data","year":2012,"authors":["Frontini, F.","Quochi, V.","Rubino, F."],"authors_source":"Frontini, Francesca; Quochi, Valeria; Rubino, Francesco","authors_cnr_name":["FRONTINI, FRANCESCA","QUOCHI, VALERIA"],"authors_cnr_id":["rp02790","rp13283"],"authors_cnr_institute":[],"abstract":"This paper describes the design of a tool for the automatic creation of multi-word lexica that is deployed as a web service and runs on automatically web-crawled data within the framework of the PANACEA platform. The main purpose of our task is to provide a (computationally \"light\") tool that creates a full high quality lexical resource of multi-word items. Within the platform, this tool is typically inserted in a work flow whose first step is automatic web-crawling. Therefore, the input data of our lexical extractor is intrinsically noisy. The paper evaluates the capacity of the tool to deal with noisy data, and in particular with texts containing a significant amount of duplicated paragraphs. The accuracy of the extraction of multi-word expressions from the original crawled corpus is compared to the accuracy of the extraction from a later \"de-duplicated\" version of the corpus. The paper shows how our method can extract with sufficiently good precision also from the original, noisy crawled data. The output of our tool is a multi-word lexicon formatted and encoded in XML according to the Lexical Mark-up Framework","keywords":["Lexical induction","multi-word extraction","web-based distributed platform","noisy data"],"pages":"","url":"http:\/\/www.kde.cs.tut.ac.jp\/~aono\/pdf\/COLING2012\/AND\/pdf\/AND04.pdf","volume":"","doi":"","editors":[],"editors_source":"","published":"Proceedings of the Sixth Workshop on Analytics for Noisy Unstructured Text Data","publisher":"ACM, Association for computing machinery (New York, USA)","issn":"","isbn":"978-1-4503-1919-5","conference_name":"AND 2012","conference_place":"New York","conference_date":"","last_updated_cnr":"2024-06-08 13:40:34","last_updated_oai":"2024-06-08 13:40:34","last_updated_www":"0000-00-00 00:00:00"},{"id":1672,"id_source":5349,"institutes":["ILC"],"type":"conference_article","type_order":7,"title":"The META-SHARE Metadata Schema for the Description of Language Resources","year":2012,"authors":["Gavrilidou, M.","Labropoulou, P.","Desipri, E.","Piperidis, S.","Papageorgiou, H.","Monachini, M.","Frontini, F.","Declerck, T.","Francopoulo, G.","Arranz, V.","Mapelli, V."],"authors_source":"Gavrilidou, Maria ; Labropoulou, Penny ; Desipri, Elina ; Piperidis, Stelio ; Papageorgiou, Haris ; Monachini, Monica ; Frontini, Francesca ; Declerck, Thierry ; Francopoulo, Gil ; Arranz, Victoria ; Mapelli, Valerie","authors_cnr_name":["MONACHINI, MONICA","FRONTINI, FRANCESCA"],"authors_cnr_id":["rp19457","rp02790"],"authors_cnr_institute":[],"abstract":"This paper presents a metadata model for the description of language resources proposed in the framework of the META-SHARE infrastructure, aiming to cover both datasets and tools\/technologies used for their processing. It places the model in the overall framework of metadata models, describes the basic principles and features of the model, elaborates on the distinction between minimal and maximal versions thereof, briefly presents the integrated environment supporting the LRs description and search and retrieval processes and concludes with work to be done in the future for the improvement of the model","keywords":["metadata","META-SHARE","LRs description"],"pages":"1090-1097","url":"http:\/\/www.lrec-conf.org\/proceedings\/lrec2012\/index.html","volume":"","doi":"","editors":[],"editors_source":"","published":"","publisher":"","issn":"","isbn":"978-2-9517408-7-7","conference_name":"The Eight International Conference on Language Resources and Evaluation (LREC'12)","conference_place":"","conference_date":"","last_updated_cnr":"2025-03-07 22:41:49","last_updated_oai":"2025-03-07 22:41:49","last_updated_www":"0000-00-00 00:00:00"},{"id":770,"id_source":119663,"institutes":["ILC"],"type":"conference_article","type_order":7,"title":"Verb interpretation for basic action types: annotation, ontology induction and creation of prototypical scenes","year":2012,"authors":["Monachini, M.","Frontini, F.","De Felice, I.","Russo, I.","Khan, F.","Gagliardi, G.","Panunzi, A."],"authors_source":"Monachini, Monica ; Frontini, Francesca ; De Felice, Irene ; Russo, Irene ; Khan, Fahad ; Gagliardi, Gloria ; Panunzi, Alessandro","authors_cnr_name":["MONACHINI, MONICA","FRONTINI, FRANCESCA","DE FELICE, IRENE","RUSSO, IRENE"],"authors_cnr_id":["rp19457","rp02790","rp05254","rp02389"],"authors_cnr_institute":[],"abstract":"In the last 20 years dictionaries and lexicographic resources such as WordNet have started to be enriched with multimodal content. Short videos depicting basic actions support the user's need (especially in second language acquisition) to fully understand the range of applicability of verbs. The IMAGACT project has among its results a repository of action verbs ontologically organised around prototypical action scenes in the form of both video recordings and 3D animations. The creation of the IMAGACT ontology, which consists in deriving action types from corpus instances of action verbs, intra and cross linguistically validating them and producing the prototypical scenes thereof, is the preliminary step for the creation of a resouce that users can browse by verb, learning how to match different action prototypes with the correct verbs in the target language. The mapping of IMAGACT types onto WordNet synsets allows for a mutual enrichment of both resources","keywords":["ontology of actions","lexical resource","3D animations"],"pages":"69-80","url":"https:\/\/iris.cnr.it\/handle\/20.500.14243\/119663","volume":"","doi":"","editors":[],"editors_source":"","published":"","publisher":"","issn":"","isbn":"","conference_name":"COLING 2012-3rd Workshop on Cognitive Aspects of the Lexicon (CogALex-III)","conference_place":"","conference_date":"","last_updated_cnr":"2024-05-28 16:54:35","last_updated_oai":"2024-05-28 16:54:35","last_updated_www":"0000-00-00 00:00:00"},{"id":156,"id_source":122911,"institutes":["ILC"],"type":"conference_article","type_order":7,"title":"IMAGACT: Deriving an Action Ontology from Spoken Corpora","year":2012,"authors":["Moneglia, M.","Gagliardi, G.","Panunzi, A.","Frontini, F.","Russo, I.","Monachini, M."],"authors_source":"Moneglia, Massimo; Gagliardi, Gloria; Panunzi, Alessandro; Frontini, Francesca; Russo, Irene; Monachini, Monica","authors_cnr_name":["FRONTINI, FRANCESCA","RUSSO, IRENE","MONACHINI, MONICA"],"authors_cnr_id":["rp02790","rp02389","rp19457"],"authors_cnr_institute":[],"abstract":"This paper presents the IMAGACT annotation infrastructure which uses both corpus-based and competence-based methods for the simultaneous extraction of a language independent Action ontology from English and Italian spontaneous speech corpora. The infrastructure relies on an innovative methodology based on images of prototypical scenes and will identify high frequency action concepts in everyday life, suitable for the implementation of an open set of languages","keywords":["Action verb","Ontology","imagery"],"pages":"42-47","url":"https:\/\/iris.cnr.it\/handle\/20.500.14243\/122911","volume":"","doi":"","editors":["Bunt, H."],"editors_source":"Bunt H.","published":"Proceedings of the Eight Joint ISO-ACL SIGSEM Workshop on Interoperable Semantic Annotation ISA-8","publisher":"","issn":"","isbn":"978-90-74029-00-1","conference_name":"Eighth Joint ISO-ACL SIGSEM Workshop on Interoperable Semantic Annotation (ISA-8)","conference_place":"","conference_date":"","last_updated_cnr":"2024-05-17 17:00:45","last_updated_oai":"2024-05-17 17:00:45","last_updated_www":"0000-00-00 00:00:00"},{"id":424,"id_source":5301,"institutes":["ILC"],"type":"conference_article","type_order":7,"title":"The IMAGACT Cross-linguistic Ontology of Action. A new infrastructure for natural language disambiguation","year":2012,"authors":["Moneglia, M.","Monachini, M.","Calabrese, O.","Panunzi, A.","Frontini, F.","Gagliardi, G.","Russo, I."],"authors_source":"Moneglia, Massimo ; Monachini, Monica ; Calabrese, Omar ; Panunzi, Alessandro ; Frontini, Francesca ; Gagliardi, Gloria ; Russo, Irene","authors_cnr_name":["MONACHINI, MONICA","FRONTINI, FRANCESCA","RUSSO, IRENE"],"authors_cnr_id":["rp19457","rp02790","rp02389"],"authors_cnr_institute":[],"abstract":"Action verbs, which are highly frequent in speech, cause disambiguation problems that are relevant to Language Technologies. This is a consequence of the peculiar way each natural language categorizes Action i. e. it is a consequence of semantic factors. Action verbs are frequently \"general\", since they extend productively to actions belonging to different ontological types. Moreover, each language categorizes action in its own way and therefore the cross-linguistic reference to everyday activities is puzzling. This paper briefly sketches the IMAGACT project, which aims at setting up a cross-linguistic Ontology of Action for grounding disambiguation tasks in this crucial area of the lexicon. The project derives information on the actual variation of action verbs in English and Italian from spontaneous speech corpora, where references to action are high in frequency. Crucially it makes use of the universal language of images to identify action types, avoiding the underdeterminacy of semantic definitions. Action concept entries are implemented as prototypic scenes; this will make it easier to extend the Ontology to other languages","keywords":["Action verbs","Ontology","Imagery"],"pages":"2606-2613","url":"http:\/\/www.lrec-conf.org\/proceedings\/lrec2012\/pdf\/428_Paper.pdf","volume":"","doi":"","editors":[],"editors_source":"","published":"","publisher":"","issn":"","isbn":"978-2-9517408-7-7","conference_name":"The Eight International Conference on Language Resources and Evaluation (LREC'12)","conference_place":"","conference_date":"","last_updated_cnr":"2025-06-12 01:45:12","last_updated_oai":"2025-06-12 01:45:12","last_updated_www":"0000-00-00 00:00:00"},{"id":1849,"id_source":122919,"institutes":["ILC"],"type":"conference_article","type_order":7,"title":"Mapping a corpusinduced ontology of action verbs on ItalWordNet","year":2012,"authors":["Moneglia, M.","Monachini, M.","Panunzi, A.","Frontini, F.","Gagliardi, G.","Russo, I."],"authors_source":"Moneglia, Massimo; Monachini, Monica; Panunzi, Alessandro; Frontini, Francesca; Gagliardi, Gloria; Russo, Irene","authors_cnr_name":["MONACHINI, MONICA","FRONTINI, FRANCESCA","RUSSO, IRENE"],"authors_cnr_id":["rp19457","rp02790","rp02389"],"authors_cnr_institute":[],"abstract":"Action verbs are the least predictable linguistic type for bilingual dictionaries and they cause major problems for NLP technologies. This is not only because of language specific phraseology, but it is rather a consequence of the peculiar way each language categorizes events. In ordinary languages the most frequent action verbs are \"general\", since they extend productively to actions belonging to different ontological types. Moreover, each language categorizes actions in its own way and therefore the cross-linguistic reference to everyday activities is puzzling. A cross-linguistic stable ontology of actions is difficult to achieve because our knowledge on the actual variation of verbs across types of actions is largely unknown. This paper briefly presents the problems and the building strategies of the IMAGACT Ontology, which aims at filling this gap, and compares some early results on a set of Italian verbs with the information contained in ItalWordNet","keywords":["action verb","ontology","image"],"pages":"219-226","url":"https:\/\/iris.cnr.it\/handle\/20.500.14243\/122919","volume":"","doi":"","editors":["Fellbaum, C.","Vossen, P."],"editors_source":"Fellbaum C., Vossen P.","published":"Proceedings of the 6th Global WordNet Conference (GWC2012)","publisher":"","issn":"","isbn":"978-80-263-0244-5","conference_name":"Global Wordnet Conference (GWC2012)","conference_place":"","conference_date":"","last_updated_cnr":"2024-05-18 13:14:26","last_updated_oai":"2024-05-18 13:14:26","last_updated_www":"0000-00-00 00:00:00"},{"id":325,"id_source":128266,"institutes":["ILC"],"type":"conference_article","type_order":7,"title":"A MWE Acquisition and Lexicon Builder Web Service","year":2012,"authors":["Quochi, V.","Frontini, F.","Rubino, F."],"authors_source":"Quochi, Valeria; Frontini, Francesca; Rubino, Francesco","authors_cnr_name":["QUOCHI, VALERIA","FRONTINI, FRANCESCA"],"authors_cnr_id":["rp13283","rp02790"],"authors_cnr_institute":[],"abstract":"This paper describes the development of a web-service tool for the automatic extraction of Multi-word expressions lexicons, which has been integrated in a distributed platform for the automatic creation of linguistic resources. The main purpose of the work described is thus to provide a (computationally \"light\") tool that produces a full lexical resource: multi-word terms\/items with relevant and useful attached information that can be used for more complex processing tasks and applications (e. g. parsing, MT, IE, query expansion, etc.). The output of our tool is a MW lexicon formatted and encoded in XML according to the Lexical Mark-up Framework. The tool is already functional and available as a service. Evaluation experiments show that the tool precision is of about 80%","keywords":["Multiword extraction","lexical resources","LMF","web services."],"pages":"2291-2306","url":"http:\/\/aclweb.org\/anthology\/C\/C12\/C12-1140.pdf","volume":"","doi":"","editors":["Kay, M.","Boitet, C."],"editors_source":"Martin Kay and Christian Boitet","published":"Proceedings of COLING 2012: Technical Papers","publisher":"Curran Associates (Red Hook, NY 12571, USA)","issn":"","isbn":"9781627483896","conference_name":"International Conference on Computational Linguistics (COLING)","conference_place":"Red Hook, NY 12571","conference_date":"","last_updated_cnr":"2024-05-12 11:34:50","last_updated_oai":"2024-05-12 11:34:50","last_updated_www":"0000-00-00 00:00:00"},{"id":951,"id_source":128261,"institutes":["ILC"],"type":"conference_article","type_order":7,"title":"Integrating NLP Tools in a Distributed Environment: A Case Study Chaining a Tagger with a Dependency Parser","year":2012,"authors":["Rubino, F.","Frontini, F.","Quochi, V."],"authors_source":"Rubino, Francesco; Frontini, Francesca; Quochi, Valeria","authors_cnr_name":["FRONTINI, FRANCESCA","QUOCHI, VALERIA"],"authors_cnr_id":["rp02790","rp13283"],"authors_cnr_institute":[],"abstract":"The present paper tackles the issue of PoS tag conversion within the framework of a distributed web service platform for the automatic creation of language resources. PoS tagging is now considered a \"solved problem\"; yet, because of the differences in the tagsets, interchange of the various PoS taggers vailable is still hampered. In this paper we describe the implementation of a PoS-tagged-corpus converter, which is needed for chaining together in a workflow the FreeLing PoS tagger for Italian and the DESR dependency parser, given that these two tools have been developed independently. The conversion problems experienced during the implementation, related to the properties of the different tagsets and of tagset conversion in general, are discussed together with the solutions adopted. Finally, the converter is evaluated by assessing the impact of conversion on the performance of the dependency parser by comparing with the outcome of the native pipeline. From this we learn that in most cases parsing errors are due to actual tagging errors, and not to conversion itself. Besides, information on accuracy loss is an important feature in a distributed environment of (NLP) services, where users need to decide which services best suit their needs","keywords":["PoS tag conversion","interoperability","NLP pipelines"],"pages":"2125-2131","url":"http:\/\/www.lrec-conf.org\/proceedings\/lrec2012\/summaries\/726.html","volume":"","doi":"","editors":["Calzolari, N.","Choukri, K.","Declerck, T.","Do\u011fan, M. U.","Maegaard, B.","Mariani, J.","Odijk, J.","Piperidis, S."],"editors_source":"Nicoletta Calzolari, Khalid Choukri, Thierry Declerck, Mehmet U?ur Do?an, Bente Maegaard, Joseph Mariani, Jan Odijk, Stelios Piperidis","published":"Proceedings of the Eight International Conference on Language Resources and Evaluation (LREC'12)","publisher":"European language resources association (ELRA) (Paris, FRA)","issn":"","isbn":"9782951740877","conference_name":"Language Resources and Evaluation Conference 2012","conference_place":"Paris","conference_date":"","last_updated_cnr":"2025-03-01 07:20:18","last_updated_oai":"2025-03-01 07:20:18","last_updated_www":"0000-00-00 00:00:00"},{"id":896,"id_source":314751,"institutes":["ILC","IIT"],"type":"conference_misc","type_order":8,"title":"Web Language Identification Testing Tool","year":2012,"authors":["Frontini, F.","Monachini, M.","N Lapolla, M.","Marchetti, A.","Abrate, M.","Bacciu, C."],"authors_source":"Frontini, F; Monachini, M; N LaPolla, M; Marchetti, A; Abrate, M; Bacciu, C","authors_cnr_name":["FRONTINI, FRANCESCA","MONACHINI, MONICA","MARCHETTI, ANDREA","ABRATE, MATTEO","BACCIU, CLARA"],"authors_cnr_id":["rp02790","rp19457","rp24107","rp02900","rp02898"],"authors_cnr_institute":[],"abstract":"Nowadays a variety of tools for automatic language identification are available. Regardless of the approach used, at least two features can be identified as crucial to evaluate the performances of such tools: the precision of the presented results and the range of languages that can be detected. In this work we shall focus on a subtask of written language identification that is important to preserve and enhance multilinguality in the Web, i. e. detecting the language of a Web page given its URL. Most specifically, the final aim is to verify to which extent under-represented languages are recognized by available tools. The main specificity of Web Language Identification (WLI) lies in the fact that often an HTML page can provide interesting extralinguistic clues (URL domain name, metadata, encoding, etc) that can enhance accuracy. We shall first provide some data and statistics on the presence of languages on the web, secondly discuss existing practices and tools for language identification according to different metrics-for instance the approaches used and the number of supported languages-and finally make some proposals on how to improve current Web Language Identifiers. We shall also present a preliminary WLI service that builds on the Google Chromium Compact Language Detector; the WLI tool allows us to test the Google n-gram based algorithm against an ad-hoc gold standard of pages in various languages. The gold standard, based on a selection of Wikipedia projects, contains samples in languages for which no automatic recognition has been attempted; it can thus be used by specialists to develop and evaluate WLI systems","keywords":["Language Identification Tools","Multilingual Web"],"pages":"1-1","url":"https:\/\/iris.cnr.it\/handle\/20.500.14243\/314751","volume":"","doi":"","editors":[],"editors_source":"","published":"","publisher":"","issn":"","isbn":"","conference_name":"W3C Workshop, Call for Participation: The Multilingual Web-The Way Ahead","conference_place":"","conference_date":"","last_updated_cnr":"2024-05-05 19:55:19","last_updated_oai":"2024-05-05 19:55:19","last_updated_www":"0000-00-00 00:00:00"},{"id":765,"id_source":130245,"institutes":["ILC","IIT"],"type":"technical_report","type_order":9,"title":"Specifiche architetturali e funzionali","year":2012,"authors":["Aliprandi, C.","Bacciu, C.","Bartolini, R.","Frontini, F.","Lapolla, N.","Marchetti, A.","Piccinonno, F.","Soru, T."],"authors_source":"Aliprandi, Carlo; Bacciu, Clara; Bartolini, Roberto; Frontini, Francesca; Lapolla, Noemi; Marchetti, Andrea; Piccinonno, Fulvio; Soru, Tiziana","authors_cnr_name":["BACCIU, CLARA","BARTOLINI, ROBERTO","FRONTINI, FRANCESCA","MARCHETTI, ANDREA"],"authors_cnr_id":["rp02898","rp00239","rp02790","rp24107"],"authors_cnr_institute":[],"abstract":"Questo documento contiene le specifiche funzionali ed architetturali del sistema GLOSS elaborate come risultato dell'obiettivo operativo 1. Tali specifiche debbono essere di riferimento per tutte le fasi di sviluppo dei vari componenti del sistema stesso e della loro integrazione in un prototipo dimostrativo. Ad una breve introduzione che richiama gli obiettivi generali del progetto, seguono: 1. La descrizione delle funzionalit\u00e0 suddivisa nelle varie fasi che compongono il flusso operativo di GLOSS. 2. La descrizione dell'architettura del sistema da realizzare nella quale si fornisce lo schema dell'integrazione dei vari componenti, il protocollo di comunicazione e memorizzazione dei dati che viene trattato pi\u00f9 nel dettaglio nel documento D1. 2 GAF-Gloss Annotation Format, e la descrizione di ciascun componente del sistema. Per sua natura, questo documento sar\u00e0 soggetto a revisione durante tutto il periodo di sviluppo del sistema. Questa prima versione deve intendersi come guida per l'implementazione ed ha lo scopo di fornire a chi partecipa a questo progetto una visione generale delle funzionalit\u00e0 di GLOSS e come queste dovranno essere integrate nel prototipo dimostratore","keywords":["GLOSS","specifiche funzionali"],"pages":"","url":"https:\/\/iris.cnr.it\/handle\/20.500.14243\/130245","volume":"","doi":"","editors":[],"editors_source":"","published":"","publisher":"","issn":"","isbn":"","conference_name":"","conference_place":"","conference_date":"","last_updated_cnr":"2024-05-07 11:12:45","last_updated_oai":"2024-05-07 11:12:45","last_updated_www":"0000-00-00 00:00:00"},{"id":1555,"id_source":129408,"institutes":["ILC"],"type":"technical_report","type_order":9,"title":"D4. 5 Final Report on the Corpus Acquisition & Annotation subsystem and its components","year":2012,"authors":["Prokopidis, P.","Papavassiliou, V.","Toral, A.","Poch Riera, M.","Frontini, F.","Rubino, F.","Thurmair, G."],"authors_source":"Prokopidis, Prokopis; Papavassiliou, Vassilis; Toral, Antonio; Poch Riera, Marc; Frontini, Francesca; Rubino, Francesco; Thurmair, Gregor","authors_cnr_name":["FRONTINI, FRANCESCA"],"authors_cnr_id":["rp02790"],"authors_cnr_institute":[],"abstract":"PANACEA WP4 targets the creation of a Corpus Acquisition and Annotation (CAA) subsystem for the acquisition and processing of monolingual and bilingual language resources (LRs). The CAA subsystem consists of tools that have been integrated as web services in the PANACEA platform of LR production. D4. 2 Initial functional prototype and documentation in T13 and D4. 4 Report on the revised Corpus Acquisition & Annotation subsystem and its components in T23 provided initial and updated documentation on this subsystem, while this deliverable presents the final documentation of the subsystem as it evolved after the third development cycle of the project. The deliverable is structured as follows. The Corpus Acquisition Component (i. e. the Focused Monolingual and Bilingual Crawlers (FMC\/FBC)) is described in section 2. The final list of tools for corpus normalization (cleaning and de-duplication) is detailed in section 3. Section 4 provides documentation on all NLP tools included in the subsystem. Due to its nature, this deliverable aggregates considerable parts of all previous WP4 deliverables. The main new additions include a) new functionalities for, among others, crawling strategy, de-duplication, and detection of parallel document pairs; and b) new NLP tools for syntactic analysis, named entity recognition, tweet processing and anonymization","keywords":["Corpus Acquisition"],"pages":"","url":"http:\/\/www.jotform.com\/uploads\/fabioaffeilc\/30222975566357\/225350067351490116\/PANACEA","volume":"","doi":"","editors":[],"editors_source":"","published":"","publisher":"","issn":"","isbn":"","conference_name":"","conference_place":"","conference_date":"","last_updated_cnr":"2024-05-14 20:37:41","last_updated_oai":"2024-05-14 20:37:41","last_updated_www":"0000-00-00 00:00:00"},{"id":1476,"id_source":130130,"institutes":["ILC"],"type":"technical_report","type_order":9,"title":"D7. 4 Third evaluation report. Evaluation of PANACEA v3 and produced resources","year":2012,"authors":["Quochi, V.","Frontini, F.","Bartolini, R.","Hamon, O.","Poch Riera, M.","Padro, M.","Bel, N.","Thurmair, G.","Toral, A.","Kamran, A."],"authors_source":"Quochi, Valeria; Frontini, Francesca; Bartolini, Roberto; Hamon, Olivier; Poch Riera, Marc; Padro, Muntsa; Bel, Nuria; Thurmair, Gregor; Toral, Antonio; Kamran, Amir","authors_cnr_name":["QUOCHI, VALERIA","FRONTINI, FRANCESCA","BARTOLINI, ROBERTO"],"authors_cnr_id":["rp13283","rp02790","rp00239"],"authors_cnr_institute":[],"abstract":"D7. 4 reports on the evaluation of the different components integrated in the PANACEA third cycle of development as well as the final validation of the platform itself. All validation and evaluation experiments follow the evaluation criteria already described in D7. 1. The main goal of WP7 tasks was to test the (technical) functionalities and capabilities of the middleware that allows the integration of the various resource-creation components into an interoperable distributed environment (WP3) and to evaluate the quality of the components developed in WP5 and WP6. The content of this deliverable is thus complementary to D8. 2 and D8. 3 that tackle advantages and usability in industrial scenarios. It has to be noted that the PANACEA third cycle of development addressed many components that are still under research. The main goal for this evaluation cycle thus is to assess the methods experimented with and their potentials for becoming actual production tools to be exploited outside research labs. For most of the technologies, an attempt was made to re-interpret standard evaluation measures, usually in terms of accuracy, precision and recall, as measures related to a reduction of costs (time and human resources) in the current practices based on the manual production of resources. In order to do so, the different tools had to be tuned and adapted to maximize precision and for some tools the possibility to offer confidence measures that could allow a separation of the resources that still needed manual revision has been attempted. Furthermore, the extension to other languages in addition to English, also a PANACEA objective, has been evaluated. The main facts about the evaluation results are now summarized","keywords":["PANACEA","evaluation","machine translation"],"pages":"","url":"https:\/\/iris.cnr.it\/handle\/20.500.14243\/130130","volume":"","doi":"","editors":[],"editors_source":"","published":"","publisher":"","issn":"","isbn":"","conference_name":"","conference_place":"","conference_date":"","last_updated_cnr":"2024-04-23 22:35:06","last_updated_oai":"2024-04-23 22:35:06","last_updated_www":"0000-00-00 00:00:00"},{"id":1535,"id_source":130143,"institutes":["ILC"],"type":"technical_report","type_order":9,"title":"D6. 2 Integrated Final Version of the Components for Lexical Acquisition","year":2012,"authors":["Rimell, L.","Bel, N.","Padr\u00f3, M.","Frontini, F.","Monachini, M.","Quochi, V."],"authors_source":"Rimell, Laura; Bel, N\u00faria; Padr\u00f3, Muntsa; Frontini, Francesca; Monachini, Monica; Quochi, Valeria","authors_cnr_name":["FRONTINI, FRANCESCA","MONACHINI, MONICA","QUOCHI, VALERIA"],"authors_cnr_id":["rp02790","rp19457","rp13283"],"authors_cnr_institute":[],"abstract":"The PANACEA project has addressed one of the most critical bottlenecks that threaten the development of technologies to support multilingualism in Europe, and to process the huge quantity of multilingual data produced annually. Any attempt at automated language processing, particularly Machine Translation (MT), depends on the availability of language-specific resources. Such Language Resources (LR) contain information about the language's lexicon, i. e. the words of the language and the characteristics of their use. In Natural Language Processing (NLP), LRs contribute information about the syntactic and semantic behaviour of words-i. e. their grammar and their meaning-which inform downstream applications such as MT. To date, many LRs have been generated by hand, requiring significant manual labour from linguistic experts. However, proceeding manually, it is impossible to supply LRs for every possible pair of European languages, textual domain, and genre, which are needed by MT developers. Moreover, an LR for a given language can never be considered complete nor final because of the characteristics of natural language, which continually undergoes changes, especially spurred on by the emergence of new knowledge domains and new technologies. PANACEA has addressed this challenge by building a factory of LRs that progressively automates the stages involved in the acquisition, production, updating and maintenance of LRs required by MT systems. The existence of such a factory will significantly cut down the cost, time and human effort required to build LRs. WP6 has addressed the lexical acquisition component of the LR factory, that is, the techniques for automated extraction of key lexical information from texts, and the automatic collation of lexical information into LRs in a standardized format. The goal of WP6 has been to take existing techniques capable of acquiring syntactic and semantic information from corpus data, improving upon them, adapting and applying them to multiple languages, and turning them into powerful and flexible techniques capable of supporting massive applications. One focus for improving the scalability and portability of lexical acquisition techniques has been to extend exiting techniques with more powerful, less \"supervised\" methods. In NLP, the amount of supervision refers to the amount of manual annotation which must be applied to a text corpus before machine learning or other techniques are applied to the data to compile a lexicon. More manual annotation means more accurate training data, and thus a more accurate LR. However, given that it is impractical from a cost and time perspective to manually annotate the vast amounts of data required for multilingual MT across domains, it is important to develop techniques which can learn from corpora with less supervision. Less supervised methods are capable of supporting both large-scale acquisition and efficient domain adaptation, even in the domains where data is scarce. Another focus of lexical acquisition in PANACEA has been the need of LR users to tune the accuracy level of LRs. Some applications may require increased precision, or accuracy, where the application requires a high degree of confidence in the lexical information used. At other times a greater level of coverage may be required, with information about more words at the expense of some degree of accuracy. Lexical acquisition in PANACEA has investigated confidence thresholds for lexical acquisition to ensure that the ultimate users of LRs can generate lexical data from the PANACEA factory at the desired level of accuracy","keywords":["Lexical Acquisition"],"pages":"","url":"http:\/\/www.panacea-lr.eu\/system\/deliverables\/PANACEA_D6.2.pdf","volume":"","doi":"","editors":[],"editors_source":"","published":"","publisher":"","issn":"","isbn":"","conference_name":"","conference_place":"","conference_date":"","last_updated_cnr":"2024-05-01 00:59:32","last_updated_oai":"2024-05-01 00:59:32","last_updated_www":"0000-00-00 00:00:00"},{"id":940,"id_source":130256,"institutes":["ILC"],"type":"technical_report","type_order":9,"title":"D6. 3 Monolingual lexica for English, Spanish and Italian tuned for a particular domain (LAB and ENV)","year":2012,"authors":["Rimell, L.","Bel, N.","Padr\u00f2, M.","Frontini, F.","Monachini, M.","Quochi, V.","Del Gratta, R."],"authors_source":"Rimell, Laura; Bel, Nuria; Padr\u00f2, Muntsa; Frontini, Francesca; Monachini, Monica; Quochi, Valeria; Del Gratta, Riccardo","authors_cnr_name":["FRONTINI, FRANCESCA","MONACHINI, MONICA","QUOCHI, VALERIA","DEL GRATTA, RICCARDO"],"authors_cnr_id":["rp02790","rp19457","rp13283","rp00284"],"authors_cnr_institute":[],"abstract":"This document presents the lexica acquired using PANACEA platform for Labour and Environment domains. The languages of the lexica are English, Spanish and Italian. The lexical information acquired depends on the language, according to the available tools in the platform","keywords":["Lexicon Acqusition"],"pages":"","url":"http:\/\/www.panacea-lr.eu\/system\/deliverables\/PANACEA_D6.3.pdf","volume":"","doi":"","editors":[],"editors_source":"","published":"","publisher":"","issn":"","isbn":"","conference_name":"","conference_place":"","conference_date":"","last_updated_cnr":"2024-03-23 00:01:57","last_updated_oai":"2024-03-23 00:01:57","last_updated_www":"0000-00-00 00:00:00"},{"id":1015,"id_source":130161,"institutes":["ILC"],"type":"technical_report","type_order":9,"title":"D6. 5 Merged dictionaries","year":2012,"authors":["Rimell, L.","Bel, N.","Padr\u00f3, M.","Frontini, F.","Monachini, M.","Quochi, V.","Del Gratta, R."],"authors_source":"Rimell, Laura; Bel, N\u00faria; Padr\u00f3, Muntsa; Frontini, Francesca; Monachini, Monica; Quochi, Valeria; Del Gratta, Riccardo","authors_cnr_name":["FRONTINI, FRANCESCA","MONACHINI, MONICA","QUOCHI, VALERIA","DEL GRATTA, RICCARDO"],"authors_cnr_id":["rp02790","rp19457","rp13283","rp00284"],"authors_cnr_institute":[],"abstract":"This document presents the merged dictionaries delivered in PANACEA. Those dictionaries result from merging already existing lexica, generally for general domain, with domain specific lexica acquired using PANACEA platform. The domain specific lexica are presented and delivered in D6. 3 and the merging repository that allowed the multilevel merging in D6. 4","keywords":["merged dictionaries","computational lexicon"],"pages":"","url":"http:\/\/www.panacea-lr.eu\/\/en\/deliverables\/list","volume":"","doi":"","editors":[],"editors_source":"","published":"","publisher":"","issn":"","isbn":"","conference_name":"","conference_place":"","conference_date":"","last_updated_cnr":"2024-03-31 13:02:55","last_updated_oai":"2024-03-31 13:02:55","last_updated_www":"0000-00-00 00:00:00"},{"id":1362,"id_source":128221,"institutes":["ILC","IIT"],"type":"misc","type_order":11,"title":"Web Language Identification Testing Tool","year":2012,"authors":["Abrate, M.","Bacciu, C.","Frontini, F.","Lapolla Mariantonietta, N.","Marchetti, A.","Monachini, M."],"authors_source":"Abrate, Matteo; Bacciu, Clara; Frontini, Francesca; Lapolla Mariantonietta, Noemi; Marchetti, Andrea; Monachini, Monica","authors_cnr_name":["ABRATE, MATTEO","BACCIU, CLARA","FRONTINI, FRANCESCA","MARCHETTI, ANDREA","MONACHINI, MONICA"],"authors_cnr_id":["rp02900","rp02898","rp02790","rp24107","rp19457"],"authors_cnr_institute":[],"abstract":"Nowadays a variety of tools for automatic language identification are available. Regardless of the approach used, at least two features can be identified as crucial to evaluate the performances of such tools: the precision of the presented results and the range of languages that can be detected. In this work we shall focus on a subtask of written language identification that is important to preserve and enhance multilinguality in the Web, i. e. detecting the language of a Web page given its URL. Most specifically, the final aim is to verify to which extent under-represented languages are recognized by available tools. The main specificity of Web Language Identification (WLI) lies in the fact that often an HTML page can provide interesting extralinguistic clues (URL domain name, metadata, encoding, etc) that can enhance accuracy. We shall first provide some data and statistics on the presence of languages on the web, secondly discuss existing practices and tools for language identification according to different metrics-for instance the approaches used and the number of supported languages-and finally make some proposals on how to improve current Web Language Identifiers. We shall also present a preliminary WLI service that builds on the Google Chromium Compact Language Detector; the WLI tool allows us to test the Google n-gram based algorithm against an adhoc gold standard of pages in various languages. The gold standard, based on a selection of Wikipedia projects, contains samples in languages for which no automatic recognition has been attempted; it can thus be used by specialists to develop and evaluate WLI systems","keywords":["Multilingual Web"],"pages":"","url":"https:\/\/iris.cnr.it\/handle\/20.500.14243\/128221","volume":"","doi":"","editors":[],"editors_source":"","published":"","publisher":"","issn":"","isbn":"","conference_name":"The Multilingual Web-the Way Ahead","conference_place":"","conference_date":"","last_updated_cnr":"2024-06-22 13:32:19","last_updated_oai":"2024-06-22 13:32:19","last_updated_www":"0000-00-00 00:00:00"},{"id":463,"id_source":214980,"institutes":["ILC"],"type":"conference_article","type_order":7,"title":"The Language Library: Many Layers, More Knowledge","year":2011,"authors":["Calzolari, N.","Del Gratta, R.","Frontini, F.","Russo, I."],"authors_source":"Calzolari, Nicoletta; DEL GRATTA, Riccardo; Frontini, Francesca; Russo, Irene","authors_cnr_name":["DEL GRATTA, RICCARDO","FRONTINI, FRANCESCA","RUSSO, IRENE"],"authors_cnr_id":["rp00284","rp02790","rp02389"],"authors_cnr_institute":[],"abstract":"In this paper we outline the general concept of the Language Library, a new initiative that has the purpose of building a huge archive of structured colletion of linguistic information. The Language Library is conceived as a community built repository and as an environment that allows language specialists to share multidimensional and multi-level annotated\/processed resources. The first steps towards its implementation are briefly sketched","keywords":["Language Resources","Language Library"],"pages":"93-97","url":"https:\/\/iris.cnr.it\/handle\/20.500.14243\/214980","volume":"","doi":"","editors":[],"editors_source":"","published":"Workshop on Language Resources, Technology and Services in the Sharing Paradigm","publisher":"","issn":"","isbn":"978-974-466-564-5","conference_name":"Workshop on Language Resources, Technology and Services in the Sharing Paradigm","conference_place":"","conference_date":"","last_updated_cnr":"2024-12-24 16:10:48","last_updated_oai":"2024-12-24 16:10:48","last_updated_www":"0000-00-00 00:00:00"},{"id":1611,"id_source":215017,"institutes":["ILC"],"type":"conference_article","type_order":7,"title":"A Metadata Schema for the Description ofLanguage Resources (LRs)","year":2011,"authors":["Frontini, F.","Monachini, M.","Gavrilidou, M.","Labropoulou, P.","Piperidis, S.","Francopoulo, G.","Arranz, V.","Mapelli, V."],"authors_source":"Frontini Francesca; Monachini Monica; Gavrilidou Maria; Labropoulou Penny; Piperidis Stelios; Francopoulo Gil; Arranz Victoria; Mapelli Valerie","authors_cnr_name":["FRONTINI, FRANCESCA","MONACHINI, MONICA"],"authors_cnr_id":["rp02790","rp19457"],"authors_cnr_institute":[],"abstract":"This paper presents the metadata schema for describing language resources (LRs) currently under development for the needs of META-SHARE, an open distributed facility for the exchange and sharing of LRs. An essential ingredient in its setup is the existence of formal and standardized LR descriptions, cornerstone of the interoperability layer of any such initiative. The description of LRs is granular and abstractive, combining the taxonomy of LRs with an inventory of a structured set of descriptive elements, of which only a minimal subset is obligatory; the schema additionally proposes recommended and optional elements. Moreover, the schema includes a set of relations catering for the appropriate inter-linking of resources. The current paper presents the main principles and features of the metadata schema, focusing on the description of text corpora and lexical \/ conceptual resources","keywords":["metadata","language resources"],"pages":"84-92","url":"https:\/\/iris.cnr.it\/handle\/20.500.14243\/215017","volume":"","doi":"","editors":[],"editors_source":"","published":"","publisher":"","issn":"","isbn":"978-974-466-564-5","conference_name":"Workshop on Language Resources, Technology and Services in the Sharing Paradigm","conference_place":"","conference_date":"","last_updated_cnr":"2024-05-15 12:31:26","last_updated_oai":"2024-05-15 12:31:26","last_updated_www":"0000-00-00 00:00:00"},{"id":927,"id_source":231385,"institutes":["ILC"],"type":"technical_report","type_order":9,"title":"The FLaReNet Databook","year":2011,"authors":["Arranz, V.","Bel, N.","Budin, G.","Caselli, T.","Choukri, K.","Del Gratta, R.","Frontini, F.","Goggi, S.","Monachini, M.","Quochi, V.","Rubino, F.","Russo, I."],"authors_source":"Arranz, V; Bel, N; Budin, G; Caselli, T; Choukri, K; Del Gratta, R; Frontini, F; Goggi, S; Monachini, M; Quochi, V; Rubino, F; Russo, I et alii","authors_cnr_name":["DEL GRATTA, RICCARDO","FRONTINI, FRANCESCA","GOGGI, SARA","MONACHINI, MONICA","QUOCHI, VALERIA"],"authors_cnr_id":["rp00284","rp02790","rp20855","rp19457","rp13283"],"authors_cnr_institute":[],"abstract":"The FLaReNet Databook is not only the collection of all the factual material collected during the activities of the project, but also a set on innovative initiatives and instruments that will remain in place for the continuous collection of such \"facts\". The purpose of the Databook is in fact, on one side, to consolidate the analyses carried out in the project and, at the same time, to set up the proper mechanisms that will enable the provision of a continuous stream of relevant factual material, also after the end of the project","keywords":["Language Resources (LRs)"],"pages":"1-8","url":"http:\/\/www.flarenet.eu\/?q=FLaReNet_Databook","volume":"","doi":"","editors":[],"editors_source":"","published":"","publisher":"","issn":"","isbn":"","conference_name":"","conference_place":"","conference_date":"","last_updated_cnr":"2024-10-03 09:37:07","last_updated_oai":"2024-10-03 09:37:07","last_updated_www":"0000-00-00 00:00:00"},{"id":526,"id_source":174745,"institutes":["ILC"],"type":"technical_report","type_order":9,"title":"Documentation and User Manual of the META-SHARE Metadata Model","year":2011,"authors":["Desipri, E.","Gavrilidou, M.","Labropoulou, P.","Piperidis, S.","Frontini, F.","Monachini, M.","Victoriaarranz","Mapelli, V.","Francopoulo, G.","Declerck, T."],"authors_source":"Elina Desipri; Maria Gavrilidou; Penny Labropoulou; Stelios Piperidis; Francesca Frontini; Monica Monachini; VictoriaArranz; Val\u00e9rie Mapelli; Gil Francopoulo; Thierry Declerck","authors_cnr_name":["FRONTINI, FRANCESCA","MONACHINI, MONICA"],"authors_cnr_id":["rp02790","rp19457"],"authors_cnr_institute":[],"abstract":"The current deliverable presents the META-SHARE metadata schema v1. 0, as implemented in the META-SHARE XSD's v1. 0 released to (META-NET and PSP partners) in July 2011 for text corpora and lexical\/conceptual resources and its supplement for audio corpora, tools and language descriptions (simplified\/refactored version) as implemented in November. It is meant to act as a user manual, providing explanations on the model contents for LRs providers and LRs curators that wish to describe their resources in accordance to it. Work on the schema is ongoing and changes\/updates to the model are constantly being made; where appropriate, some changes that are already under way are documented in this deliverable","keywords":["Language resources","metadata","standards"],"pages":"150","url":"https:\/\/iris.cnr.it\/handle\/20.500.14243\/174745","volume":"","doi":"","editors":[],"editors_source":"","published":"","publisher":"","issn":"","isbn":"","conference_name":"","conference_place":"","conference_date":"","last_updated_cnr":"2024-03-23 12:28:07","last_updated_oai":"2024-03-23 12:28:07","last_updated_www":"0000-00-00 00:00:00"},{"id":2124,"id_source":174795,"institutes":["ILC"],"type":"technical_report","type_order":9,"title":"KYOTO-LMF WordNet Representation Format","year":2011,"authors":["Monachini, M.","Frontini, F.","Soria, C."],"authors_source":"Monica Monachini; Francesca Frontini; Claudia Soria","authors_cnr_name":["MONACHINI, MONICA","FRONTINI, FRANCESCA","SORIA, CLAUDIA"],"authors_cnr_id":["rp19457","rp02790","rp17652"],"authors_cnr_institute":[],"abstract":"The format described in the following pages is the final revised proposal for representing wordnets inside the Kyoto project (henceforth \"Kyoto-LMF wordnet format\"). The reference model is Lexical Markup Framework (LMF), version 16, probably one of the most widely recognized standards for the representation of NLP lexicons. The goals of LMF are to provide a common model for the creation and use of such lexical resources, to manage the exchange of data between and among them, and to enable the merging of a large number of individual resources to form extensive global electronic respurces. LMF was specifically designed to accomodate as many models of lexical representations as possible. Purposefully, it is designed as a mea-model, i. e a high-level specification for lexical resources defining the structural constraints of a lexicon","keywords":["Wordnets","LMF","ISO","Representation formats","standards"],"pages":"32","url":"https:\/\/iris.cnr.it\/handle\/20.500.14243\/174795","volume":"","doi":"","editors":[],"editors_source":"","published":"","publisher":"","issn":"","isbn":"","conference_name":"","conference_place":"","conference_date":"","last_updated_cnr":"2024-05-12 11:34:31","last_updated_oai":"2024-05-12 11:34:31","last_updated_www":"0000-00-00 00:00:00"},{"id":36,"id_source":290521,"institutes":["ILC"],"type":"technical_report","type_order":9,"title":"WP-4. 4: Report on the revised Corpus Acquisition & Annotation subsystem and its components","year":2011,"authors":["Prokopidis, P.","Papavassiliou, V.","Toral, A.","Poch Riera, M.","Frontini, F.","Rubino, F.","Thurmair, G."],"authors_source":"Prokopidis, Prokopis; Papavassiliou, Vassilis; Toral, Antonio; Poch Riera, Marc; Frontini, Francesca; Rubino, Francesco; Thurmair, Gregor","authors_cnr_name":["FRONTINI, FRANCESCA"],"authors_cnr_id":["rp02790"],"authors_cnr_institute":[],"abstract":"","keywords":["corpus acquisition","corpus annotation"],"pages":"","url":"http:\/\/www.panacea-lr.eu\/system\/deliverables\/PANACEA_D4.4.pdf","volume":"","doi":"","editors":[],"editors_source":"","published":"","publisher":"","issn":"","isbn":"","conference_name":"","conference_place":"","conference_date":"","last_updated_cnr":"2024-04-05 10:47:27","last_updated_oai":"2024-04-05 10:47:27","last_updated_www":"0000-00-00 00:00:00"},{"id":119,"id_source":290522,"institutes":["ILC"],"type":"technical_report","type_order":9,"title":"WP-4. 5: Final Report on the Corpus Acquisition & Annotation subsystem and its components","year":2011,"authors":["Prokopidis, P.","Papavassiliou, V.","Toral, A.","Riera, M. P.","Frontini, F.","Rubino, F.","Thurmair, G."],"authors_source":"Prokopis Prokopidis; Vassilis Papavassiliou; Antonio Toral; Marc Poch Riera; Francesca Frontini; Francesco Rubino; Gregor Thurmair","authors_cnr_name":["FRONTINI, FRANCESCA"],"authors_cnr_id":["rp02790"],"authors_cnr_institute":[],"abstract":"","keywords":["corpus acquisition","corpus annotation"],"pages":"","url":"http:\/\/www.panacea-lr.eu\/system\/deliverables\/PANACEA_D4.5.pdf","volume":"","doi":"","editors":[],"editors_source":"","published":"","publisher":"","issn":"","isbn":"","conference_name":"","conference_place":"","conference_date":"","last_updated_cnr":"2024-05-14 19:19:59","last_updated_oai":"2024-05-14 19:19:59","last_updated_www":"0000-00-00 00:00:00"},{"id":1679,"id_source":174121,"institutes":["ILC"],"type":"technical_report","type_order":9,"title":"KyotoCore: integrated system for knowledge mining from text","year":2011,"authors":["Vossen, P.","Bosma, W.","Rigau, G.","Agirre, E.","Soroa, A.","Aliprandi, C.","De Jonge, J.","Hielkema, F.","Monachini, M.","Bartolini, R.","Frontini, F."],"authors_source":"Vossen, Piek; Bosma, Wauter; Rigau, German; Agirre, Eneko; Soroa, Aitor; Aliprandi, Carlo; de Jonge, Joost; Hielkema, Feikje; Monachini, Monica; Bartolini, Roberto; Frontini, Francesca","authors_cnr_name":["MONACHINI, MONICA","BARTOLINI, ROBERTO","FRONTINI, FRANCESCA"],"authors_cnr_id":["rp19457","rp00239","rp02790"],"authors_cnr_institute":[],"abstract":"In this deliverable, we describe KyotoCore, an integrated system for applying text mining. We describe the software architecture of KyotoCore, the single modules and the process flows. Finally, we describe a use case where we apply the complete process toan English database on estuaries","keywords":["Knowledge and text mining software"],"pages":"56","url":"https:\/\/iris.cnr.it\/handle\/20.500.14243\/174121","volume":"","doi":"","editors":[],"editors_source":"","published":"","publisher":"","issn":"","isbn":"","conference_name":"","conference_place":"","conference_date":"","last_updated_cnr":"2024-06-07 15:28:12","last_updated_oai":"2024-06-07 15:28:12","last_updated_www":"0000-00-00 00:00:00"},{"id":2034,"id_source":217962,"institutes":["ILC"],"type":"misc","type_order":11,"title":"Towards interfacing lexical and ontological resources","year":2011,"authors":["Frontini, F.","Monachini, M."],"authors_source":"Francesca Frontini; Monica Monachini","authors_cnr_name":["FRONTINI, FRANCESCA","MONACHINI, MONICA"],"authors_cnr_id":["rp02790","rp19457"],"authors_cnr_institute":[],"abstract":"During the last two decades, the Computational Linguistics community has dedicated considerable effort to the research and development Lexical Resources (LRs), especially Computational Lexicons. These LRs, even though belonging to different linguistic approaches and theories, share a common element; all of them contain, explicitly or implicitly, an ontology as the means of organizing their structure","keywords":["language resources","ontologies"],"pages":"26","url":"https:\/\/iris.cnr.it\/handle\/20.500.14243\/217962","volume":"","doi":"","editors":[],"editors_source":"","published":"","publisher":"","issn":"","isbn":"","conference_name":"ONTOLOGIES AND LEXICAL SEMANTICS","conference_place":"","conference_date":"","last_updated_cnr":"2024-05-29 09:45:46","last_updated_oai":"2024-05-29 09:45:46","last_updated_www":"0000-00-00 00:00:00"},{"id":1455,"id_source":134822,"institutes":["ILC"],"type":"book_chapter","type_order":4,"title":"From Pattern Dictionary to Patternbank","year":2010,"authors":["Jezek, E.","Frontini, F."],"authors_source":"Jezek E.; Frontini F.","authors_cnr_name":["FRONTINI, FRANCESCA"],"authors_cnr_id":["rp02790"],"authors_cnr_institute":[],"abstract":"","keywords":["Ontology. Computational Semantics"],"pages":"215-237","url":"https:\/\/iris.cnr.it\/handle\/20.500.14243\/134822","volume":"","doi":"","editors":["De Schryver, G. M."],"editors_source":"Gilles-Maurice de Schryver","published":"A Way with Words: Recent Advances in Lexical Theory and Analysis","publisher":"","issn":"","isbn":"","conference_name":"","conference_place":"","conference_date":"","last_updated_cnr":"2024-06-19 07:26:11","last_updated_oai":"2024-06-19 07:26:11","last_updated_www":"0000-00-00 00:00:00"},{"id":1660,"id_source":106762,"institutes":["ILC"],"type":"misc","type_order":11,"title":"Statistical profiling of Italian L2 texts: competence and native language","year":2010,"authors":["Frontini, F."],"authors_source":"Frontini, F","authors_cnr_name":["FRONTINI, FRANCESCA"],"authors_cnr_id":["rp02790"],"authors_cnr_institute":[],"abstract":"","keywords":["Text categorization"],"pages":"","url":"https:\/\/iris.cnr.it\/handle\/20.500.14243\/106762","volume":"","doi":"","editors":[],"editors_source":"","published":"","publisher":"","issn":"","isbn":"","conference_name":"20th Annual Conference of the European Second Language Association","conference_place":"","conference_date":"","last_updated_cnr":"2024-06-22 13:31:41","last_updated_oai":"2024-06-22 13:31:41","last_updated_www":"0000-00-00 00:00:00"}]