
@article{koubarakis_ai_2018,
	title = {{AI} in Greece: The Case of Research on Linked Geospatial Data},
	volume = {39},
	url = {https://www.aaai.org/ojs/index.php/aimagazine/article/view/2801},
	pages = {91--96},
	number = {2},
	journaltitle = {{AI} Magazine},
	author = {Koubarakis, Manolis and Vouros, George A. and Chalkiadakis, Georgios and Plagianakos, Vassilis P. and Tjortjis, Christos and Kavallieratou, Ergina and Vrakas, Dimitris and Mavridis, Nikolaos and Petasis, Georgios and Blekas, Konstantinos and Krithara, Anastasia},
	date = {2018},
}

@inproceedings{papachristopoulos_introducing_2018,
	location = {Chania, Greece},
	title = {Introducing Sentiment Analysis for the Evaluation of Library's Services Effectiveness},
	booktitle = {Proceedings of the 10th Qualitative and Quantitative Methods in Libraries International Conference ({QQML}2018)},
	author = {Papachristopoulos, Leonidas and Ampatzoglou, Pantelis and Seferli, Ioanna and Zafeiropoulou, Andriani and Petasis, Georgios},
	date = {2018-05-22},
}

@inproceedings{petasis_yourdatastories_2017,
	location = {Thessaloniki, Greece},
	title = {{YourDataStories}: Transparency and Corruption Fighting through Data Interlinking and Visual Exploration},
	booktitle = {Proceedings of the Data Economy Workshop, 4th International Conference on Internet Science ({INSCI} 2017)},
	author = {Petasis, Georgios and Triantafillou, Anna and Karstens, Eric},
	date = {2017-11},
}

@inproceedings{ferrara_unsupervised_2017,
	location = {Copenhagen, Denmark},
	title = {Unsupervised Detection of Argumentative Units though Topic Modeling Techniques},
	url = {http://www.aclweb.org/anthology/W17-5113},
	abstract = {In this paper we present a new unsupervised approach, "Attraction to Topics" – A2T , for the detection of argumentative units, a sub-task of argument mining. Motivated by the importance of topic identification in manual annotation, we examine whether topic modeling can be used for performing unsupervised detection of argumentative sentences, and to what extend topic modeling can be used to classify sentences as claims and premises. Preliminary evaluation results suggest that topic information can be successfully used for the detection of argumentative sentences, at least for corpora used for evaluation. Our approach has been evaluated on two English corpora, the first of which contains 90 persuasive essays, while the second is a collection of 340 documents from user generated content.},
	pages = {97--107},
	booktitle = {Proceedings of the 4th Workshop on Argument Mining, 2017 Conference on Empirical Methods in Natural Language Processing ({EMNLP} 2017)},
	publisher = {Association for Computational Linguistics},
	author = {Ferrara, Alfio and Montanelli, Stefano and Petasis, Georgios},
	date = {2017-09},
}

@inproceedings{petasis_identifying_2016,
	location = {Berlin, Germany},
	title = {Identifying Argument Components through {TextRank}},
	url = {http://aclweb.org/anthology/W/W16/W16-2811.pdf},
	pages = {56--66},
	booktitle = {Proceedings of the 3rd Workshop on Argument Mining ({ArgMining}2016), 54th Annual Meeting of the Association for Computational Linguistics ({ACL} 2016)},
	publisher = {Association for Computational Linguistics},
	author = {Petasis, Georgios and Karkaletsis, Vangelis},
	date = {2016-08},
}

@inproceedings{katakis_clarin-web-based_2016,
	title = {{CLARIN}-{EL} Web-based Annotation Tool},
	url = {http://www.lrec-conf.org/proceedings/lrec2016/summaries/990.html},
	booktitle = {Proceedings of the Tenth International Conference on Language Resources and Evaluation {LREC} 2016, Portorož, Slovenia, May 23-28, 2016.},
	publisher = {European Language Resources Association ({ELRA})},
	author = {Katakis, Ioannis Manousos and Petasis, Georgios and Karkaletsis, Vangelis},
	editor = {Calzolari, Nicoletta and Choukri, Khalid and Declerck, Thierry and Goggi, Sara and Grobelnik, Marko and Maegaard, Bente and Mariani, Joseph and Mazo, Hélène and Moreno, Asunción and Odijk, Jan and Piperidis, Stelios},
	date = {2016},
}

@article{goudas_argument_2015,
	title = {Argument Extraction from News, Blogs, and the Social Web},
	volume = {24},
	url = {http://www.worldscientific.com/doi/abs/10.1142/S0218213015400242},
	doi = {10.1142/S0218213015400242},
	pages = {1540024},
	number = {5},
	journaltitle = {International Journal on Artificial Intelligence Tools},
	author = {Goudas, Theodosis and Louizos, Christos and Petasis, Georgios and Karkaletsis, Vangelis},
	date = {2015},
	note = {\_eprint: http://www.worldscientific.com/doi/pdf/10.1142/S0218213015400242},
}

@inproceedings{krithara_predicting_2015,
	title = {Predicting Sentiment using Tranfer Learning},
	abstract = {A new transfer learning method is presented in this paper, addressing the task of sentiment analysis across domains.The proposed approach is a transfer variant of the Probabilistic Latent Semantic Analysis ({PLSA}) model that we name {KLIEP}-{PLSA}. The approach captures the difference of the tributions between the different domains. We perform experiments over well known datasets and show the promising results that we obtained new method.},
	booktitle = {Workshop on Replicability and Reproducibility in Natural Language Processing: adaptive methods, resources and software at {IJCAI} 2015 ({AdaptiveNLP} 2015)},
	author = {Krithara, Anastasia and Giannakopoulos, George and Paliouras, George and Petasis, George and Karkaletsis, Vangelis},
	date = {2015},
	keywords = {{KLIEP}-{PLSA}, {PLSA}, sentiment analysis, transfer learning},
}

@inproceedings{sardianos_argument_2015,
	location = {Denver, {CO}},
	title = {Argument Extraction from News},
	url = {http://www.aclweb.org/anthology/W15-0508},
	pages = {56--66},
	booktitle = {Proceedings of the 2nd Workshop on Argumentation Mining},
	publisher = {Association for Computational Linguistics},
	author = {Sardianos, Christos and Katakis, Ioannis Manousos and Petasis, Georgios and Karkaletsis, Vangelis},
	date = {2015-06},
}

@inproceedings{petasis_annotating_2014,
	title = {Annotating Arguments: The {NOMAD} Collaborative Annotation Tool},
	pages = {1930--1937},
	booktitle = {Proceedings of the Ninth International Conference on Language Resources and Evaluation ({LREC}-2014), Reykjavik, Iceland, May 26-31, 2014},
	publisher = {European Language Resources Association ({ELRA})},
	author = {Petasis, Georgios},
	editor = {Calzolari, Nicoletta and Choukri, Khalid and Declerck, Thierry and Loftsson, Hrafn and Maegaard, Bente and Mariani, Joseph and Moreno, Asunción and Odijk, Jan and Piperidis, Stelios},
	date = {2014},
}

@inproceedings{petasis_ellogon_2014,
	title = {The Ellogon Pattern Engine: Context-free Grammars over Annotations},
	pages = {2460--2465},
	booktitle = {Proceedings of the Ninth International Conference on Language Resources and Evaluation ({LREC}-2014), Reykjavik, Iceland, May 26-31, 2014},
	publisher = {European Language Resources Association ({ELRA})},
	author = {Petasis, Georgios},
	editor = {Calzolari, Nicoletta and Choukri, Khalid and Declerck, Thierry and Loftsson, Hrafn and Maegaard, Bente and Mariani, Joseph and Moreno, Asunción and Odijk, Jan and Piperidis, Stelios},
	date = {2014},
}

@inproceedings{kiomourtzis_nomad_2014,
	title = {{NOMAD}: Linguistic Resources and Tools Aimed at Policy Formulation and Validation},
	pages = {3464--3470},
	booktitle = {Proceedings of the Ninth International Conference on Language Resources and Evaluation ({LREC}-2014), Reykjavik, Iceland, May 26-31, 2014},
	publisher = {European Language Resources Association ({ELRA})},
	author = {Kiomourtzis, George and Giannakopoulos, George and Petasis, Georgios and Karampiperis, Pythagoras and Karkaletsis, Vangelis},
	editor = {Calzolari, Nicoletta and Choukri, Khalid and Declerck, Thierry and Loftsson, Hrafn and Maegaard, Bente and Mariani, Joseph and Moreno, Asunción and Odijk, Jan and Piperidis, Stelios},
	date = {2014},
}

@incollection{goudas_argument_2014,
	location = {Cham},
	title = {Argument Extraction from News, Blogs, and Social Media},
	isbn = {978-3-319-07064-3},
	url = {http://dx.doi.org/10.1007/978-3-319-07064-3_23},
	pages = {287--299},
	booktitle = {Artificial Intelligence: Methods and Applications: 8th Hellenic Conference on {AI}, {SETN} 2014, Ioannina, Greece, May 15-17, 2014. Proceedings},
	publisher = {Springer International Publishing},
	author = {Goudas, Theodosis and Louizos, Christos and Petasis, Georgios and Karkaletsis, Vangelis},
	editor = {Likas, Aristidis and Blekas, Konstantinos and Kalles, Dimitris},
	date = {2014},
	doi = {10.1007/978-3-319-07064-3_23},
}

@inproceedings{petasis_sentiment_2014,
	title = {Sentiment Analysis for Reputation Management: Mining the Greek Web},
	volume = {8445},
	isbn = {978-3-319-07063-6 978-3-319-07064-3},
	series = {Lecture Notes in Computer Science},
	pages = {327--340},
	booktitle = {Artificial Intelligence: Methods and Applications - 8th Hellenic Conference on {AI}, {SETN} 2014, Ioannina, Greece, May 15-17, 2014. Proceedings},
	publisher = {Springer},
	author = {Petasis, Georgios and Spiliotopoulos, Dimitris and Tsirakis, Nikos and Tsantilas, Panayotis},
	editor = {Likas, Aristidis and Blekas, Konstantinos and Kalles, Dimitris},
	date = {2014},
}

@inproceedings{petasis_large-scale_2013,
	location = {Darmstadt, Germany},
	title = {Large-scale Sentiment Analysis for Reputation Management},
	abstract = {Harvesting the web and social web data is a meticulous and complex task. Applying the results to a successful business case such as brand monitoring requires high precision and recall for the opinion mining and entity recognition tasks. This work reports on the integrated platform of a state of the art Named-entity Recognition and Classification ({NERC}) system and opinion mining methods for a Software-as-a-Service ({SaaS}) approach on a fully automatic service for brand monitoring for the Greek language. The service has been successfully deployed to the biggest search engine in Greece powering the large-scale linguistic and sentiment analysis of about 80.000 resources per hour.},
	booktitle = {Proceedings of the 2nd Workshop on Practice and Theory of Opinion Mining and Sentiment Analysis ({PATHOS}-2013)},
	author = {Petasis, Georgios and Spiliotopoulos, Dimitrios and Tsirakis, Nikos and Tsantilas, Panayiotis},
	editor = {Gindl, Stefan and Remus, Robert and Wiegand, Michael},
	date = {2013-09-23},
}

@inproceedings{petasis_structuring_2013,
	location = {Graz, Austria},
	title = {Structuring the Blogosphere on News from Traditional Media},
	pages = {608--617},
	booktitle = {On the Move to Meaningful Internet Systems: {OTM} 2013 Workshops - Confederated International Workshops: {OTM} Academy, {OTM} Industry Case Studies Program, {ACM}, {EI}2N, {ISDE}, {META}4eS, {ORM}, {SeDeS}, {SINCOM}, {SMS}, and {SOMOCO} 2013},
	author = {Petasis, Georgios},
	date = {2013-09-09},
}

@inproceedings{petasis_boemie_2013,
	location = {A Corunna, Spain},
	title = {{BOEMIE}: Reasoning-based Information Extraction},
	abstract = {This paper presents a novel approach for exploiting an ontology in an ontology-based information extraction system, which substitutes part of the extraction process with reasoning, guided by a set of automatically acquired rules.},
	pages = {60--75},
	author = {Petasis, Georgios and Möller, Ralf and Karkaletsis, Vangelis},
	date = {2013-09-15},
}

@incollection{petasis_new_2012,
	location = {Brno, Czech Republic},
	title = {A New Annotation Tool for Aligned Bilingual Corpora},
	volume = {7499},
	isbn = {978-3-642-32789-6},
	url = {http://www.ellogon.org/petasis/bibliography/TSD2012/tsd450.pdf},
	series = {Lecture Notes in Computer Science},
	abstract = {This paper presents a new annotation tool for aligned bilingual corpora, which allows the annotation of a wide range of information, ranging from information about words (such as part-of-speech tags or named-entities) to quite complex annotation schemas involving links between aligned segments, such as co-reference or translation equivalence between aligned segments in the two languages. The annotation tool is implemented as a component of the Ellogon language engineering platform, exploiting its extensive annotation engine, its cross-platform abilities and its linguistic processing components, if such a need arises. The new annotation tool is distributed with an open source license ({LGPL}), as part of the Ellogon language engineering platform.},
	pages = {95--104},
	booktitle = {Text, Speech and Dialogue},
	publisher = {Springer Berlin Heidelberg},
	author = {Petasis, Georgios and Tsoumari, Mara},
	editor = {Sojka, Petr and Horák, Aleš and Kopeček, Ivan and Pala, Karel},
	date = {2012-09-03},
	doi = {10.1007/978-3-642-32790-2_11},
	keywords = {adaptable annotation schemas, Annotation tools, collaborative annotation},
}

@inproceedings{petasis_sync3_2012,
	location = {Istanbul, Turkey},
	title = {The {SYNC}3 Collaborative Annotation Tool},
	url = {http://www.ellogon.org/petasis/bibliography/LREC2012/LREC2012-700.pdf},
	abstract = {The huge amount of the available information in the Web creates the need for effective information extraction systems that are able to produce metadata that satisfy user's information needs. The development of such systems, in the majority of cases, depends on the availability of an appropriately annotated corpus in order to learn or evaluate extraction models. The production of such corpora can be significantly facilitated by annotation tools, which provide user-friendly facilities and enable annotators to annotate documents according to a predefined annotation schema. However, the construction of annotation tools that operate in a distributed environment is a challenging task: the majority of these tools are implemented as Web applications, having to cope with the capabilities offered by browsers. This paper describes the {SYNC}3 collaborative annotation tool, which implements an alternative architecture: it remains a desktop application, fully exploiting the advantages of desktop applications, but provides collaborative annotation through the use of a centralised server for storing both the documents and their metadata, and instance messaging protocols for communicating events among all annotators. The annotation tool is implemented as a component of the Ellogon language engineering platform, exploiting its extensive annotation engine, its cross-platform abilities and its linguistic processing components, if such a need arises. Finally, the {SYNC}3 annotation tool is distributed with an open source license, as part of the Ellogon platform.},
	pages = {363--370},
	booktitle = {Proceedings of the 8th International Conference on Language Resources and Evaluation, {LREC} 2012},
	publisher = {European Language Resources Association},
	author = {Petasis, Georgios},
	date = {2012-05},
	keywords = {adaptable annotation schemas, collaborative annotation, annotation tools},
}

@incollection{iosif_ontology-based_2012,
	location = {Hershey, {PA}, {USA}},
	title = {Ontology-Based Information Extraction under a Bootstrapping Approach},
	isbn = {978-1-4666-0188-8},
	url = {http://www.igi-global.com/chapter/ontology-based-information-extraction-under/63896},
	abstract = {The authors present an ontology-based information extraction process, which operates in a bootstrapping framework. The novelty of this approach lies in the continuous semantics extraction from textual content in order to evolve the underlying ontology, while the evolved ontology enhances in turn the information extraction mechanism. This process was implemented in the context of the R\&D project {BOEMIE}. The {BOEMIE} system was evaluated on the athletics domain.},
	pages = {1--21},
	booktitle = {Semi-Automatic Ontology Development: Processes and Resources},
	publisher = {{IGI} Global},
	author = {Iosif, Elias and Petasis, Georgios and Karkaletsis, Vangelis},
	editor = {Maria Teresa Pazienza, Armando Stellato},
	date = {2012-04},
	doi = {10.4018/978-1-4666-0188-8.ch001},
	note = {Section: 1},
}

@inproceedings{sarris_system_2011,
	location = {Florence, Italy},
	title = {A System for Synergistically Structuring News Content from Traditional Media and the Blogosphere},
	isbn = {978-1-905824-27-4},
	url = {http://www.ellogon.org/petasis/bibliography/eChallenges2011/echallenges_ref_81_doc_7322.pdf},
	abstract = {News and social media are emerging as a dominant source of information for numerous applications. However, their vast unstructured content present challenges to efficient extraction of such information. In this paper, we present the {SYNC}3 system that aims to intelligently structure content from both traditional news media and the blogosphere. To achieve this goal, {SYNC}3 incorporates innovative algorithms that first model news media content statistically, based on fine clustering of articles into so-called "news events". Such models are then adapted and applied to the blogosphere domain, allowing its content to map to the traditional news domain. Furthermore, appropriate algorithms are employed to extract news event labels and relations between events, in order to efficiently present news content to the system end users.},
	booktitle = {{eChallenges} e-2011 Conference Proceedings},
	publisher = {{IIMC} International Information Management Corporation},
	author = {Sarris, Nikos and Potamianos, Gerasimos and Renders, Jean-Michel and Grover, Claire and Karstens, Eric and Kallipolitis, Leonidas and Tountopoulos, Vasilis and Petasis, Georgios and Krithara, Anastasia and Gallé, Matthias and Jacquet, Guillaume and Alex, Beatrice and Tobin, Richard and Bounegru, Liliana},
	editor = {Cunningham, Paul and Cunningham, Miriam},
	date = {2011-10-26},
}

@inproceedings{tsoumari_coreference_2011,
	title = {Coreference Annotator - A new annotation tool for aligned bilingual corpora},
	url = {http://www.aclweb.org/anthology/W11-4307},
	abstract = {This paper presents the main features of an annotation tool, the Coreference Annotator, which manages bilingual corpora consisting of aligned texts that can be grouped in collections and subcollections according to their topics and discourse. The tool allows the manual annotation of certain linguistic items in the source text and their translation equivalent in the target text, by entering useful information about these items based on their context.},
	pages = {43--52},
	booktitle = {Proceedings of the Second Workshop on Annotation and Exploitation of Parallel Corpora ({AEPC} 2), in 8th International Conference on Recent Advances in Natural Language Processing ({RANLP} 2011)},
	author = {Tsoumari, Mara and Petasis, Georgios},
	date = {2011-09-15},
}

@inproceedings{petasis_unsupervised_2011,
	location = {Hissar, Bulgaria},
	title = {Unsupervised Domain Adaptation based on Text Relatedness},
	url = {http://aclweb.org/anthology/R11-1107},
	abstract = {In this paper an unsupervised approach to do-main adaptation is presented, which exploits external knowledge sources in order to port a classification model into a new thematic do-main. Our approach extracts a new feature set from documents of the target domain, and tries to align the new features to the original ones, by exploiting text relatedness from external knowledge sources, such as {WordNet}. The approach has been evaluated on the task of document classification, involving the classification of newsgroup postings into 20 news groups.},
	pages = {733--739},
	booktitle = {Proceedings of the International Conference Recent Advances in Natural Language Processing 2011},
	publisher = {{RANLP} 2011 Organising Committee},
	author = {Petasis, Georgios},
	date = {2011-09-12},
}

@thesis{petasis_machine_2011,
	title = {Machine Learning in Natural Language Processing},
	url = {http://www.ellogon.org/petasis/bibliography/Petasis/Ph.D.Thesis-GeorgiosPetasis.pdf},
	abstract = {This thesis examines the use of machine learning techniques in various tasks of natural language processing, mainly for the task of information extraction from texts. The objectives are the improvement of adaptability of information extraction systems to new thematic domains (or even languages), and the improvement of their performance using as fewer resources (either linguistic or human) as possible. This thesis has examined two main axes: a) the research and assessment of existing algorithms of machine learning mainly in the stages of linguistic pre-processing (such as part of speech tagging) and named-entity recognition, and b) the creation of a new machine learning algorithm and its assessment on synthetic data, as well as in real world data from the task of relation extraction between named entities. This new algorithm belongs to the category of inductive grammar learning, and can infer context free grammars from positive examples only.},
	institution = {Department of Informatics and Telecommunications, University of Athens},
	type = {phdthesis},
	author = {Petasis, Georgios},
	date = {2011-07-01},
	keywords = {grammatical inference, information extraction, machine learning},
}

@incollection{karkaletsis_ontology_2011,
	title = {Ontology Based Information Extraction from Text},
	volume = {6050},
	isbn = {978-3-642-20794-5},
	url = {http://dx.doi.org/10.1007/978-3-642-20795-2_4},
	series = {Lecture Notes in Computer Science},
	abstract = {Information extraction systems employ ontologies as a means to describe formally the domain knowledge exploited by these systems for their operation. The aim of this survey is to study the contribution of ontologies to information extraction systems. We believe that this will help towards specifying a concrete methodology for ontology based information extraction exploiting all levels of ontological knowledge, from domain entities for named entity recognition, to the use of conceptual hierarchies for pattern generalization, to the use of properties and non-taxonomic relations for pattern acquisition, and finally to the use of the domain model itself for integrating extracted entities and instances of relations, as well as for discovering implicit information and detecting inconsistencies.},
	pages = {89--109},
	booktitle = {Knowledge-Driven Multimedia Information Extraction and Ontology Evolution},
	publisher = {Springer Berlin / Heidelberg},
	author = {Karkaletsis, Vangelis and Fragkou, Pavlina and Petasis, Georgios and Iosif, Elias},
	editor = {Paliouras, Georgios and Spyropoulos, Constantine D. and Tsatsaronis, George},
	date = {2011},
	doi = {10.1007/978-3-642-20795-2_4},
}

@incollection{petasis_ontology_2011,
	title = {Ontology Population and Enrichment: State of the Art},
	volume = {6050},
	isbn = {978-3-642-20794-5},
	url = {http://dx.doi.org/10.1007/978-3-642-20795-2_6},
	series = {Lecture Notes in Computer Science},
	abstract = {Ontology learning is the process of acquiring (constructing or integrating) an ontology (semi-) automatically. Being a knowledge acquisition task, it is a complex activity, which becomes even more complex in the context of the {BOEMIE} project, due to the management of multimedia resources and the multi-modal semantic interpretation that they require. The purpose of this chapter is to present a survey of the most relevant methods, techniques and tools used for the task of ontology learning. Adopting a practical perspective, an overview of the main activities involved in ontology learning is presented. This breakdown of the learning process is used as a basis for the comparative analysis of existing tools and approaches. The comparison is done along dimensions that emphasize the particular interests of the {BOEMIE} project. In this context, ontology learning in {BOEMIE} is treated and compared to the state of the art, explaining how {BOEMIE} addresses problems observed in existing systems and contributes to issues that are not frequently considered by existing approaches.},
	pages = {134--166},
	booktitle = {Knowledge-Driven Multimedia Information Extraction and Ontology Evolution},
	publisher = {Springer Berlin / Heidelberg},
	author = {Petasis, Georgios and Karkaletsis, Vangelis and Paliouras, Georgios and Krithara, Anastasia and Zavitsanos, Elias},
	editor = {Paliouras, Georgios and Spyropoulos, Constantine D. and Tsatsaronis, George},
	date = {2011},
	doi = {10.1007/978-3-642-20795-2_4},
}

@inproceedings{petasis_tkdnd_2010,
	location = {Hilton Suites Chicago/Oakbrook Terrace, 10 Drury Lane, Oakbrook Terrace, Illinois, United States 60181},
	title = {{TkDND}: a cross-platform drag'n'drop package},
	url = {http://www.ellogon.org/petasis/bibliography/Tcl2010/TkDND.pdf},
	abstract = {This paper is about {TkDND}, a Tcl/Tk extension that aims to add cross-application drag and drop support to Tk, for popular operating systems, such as Microsoft Windows, Apple {OS} X and {GNU}/Linux. Being in its second rewrite, {TkDND} 2.x has a stable implementation for Windows and {OS} X, while support for Linux and the {XDND} protocol is still under development.},
	booktitle = {Proceedings of the 17th Annual Tcl/Tk Conference (Tcl 2010)},
	author = {Petasis, Georgios},
	date = {2010-10-11},
}

@inproceedings{petasis_ellogon_2010,
	location = {Hilton Suites Chicago/Oakbrook Terrace, 10 Drury Lane, Oakbrook Terrace, Illinois, United States 60181},
	title = {Ellogon and the challenge of threads},
	url = {http://www.ellogon.org/petasis/bibliography/Tcl2010/EllogonAndThreads.pdf},
	abstract = {This paper is about the Ellogon language engineering platform, and the challenges faced in modernising it, in order to better exploit contemporary hardware. Ellogon is an open-source infrastructure, specialised in natural language processing. Following a data model that closely resembles {TIPSTER}, Ellogon can be used either as an autonomous application, offering a graphical user interface, or it can be embedded in a C/C++ application as a library. Ellogon has been implemented in C/C++ and Tcl/Tk: in fact Ellogon is a vanilla Tcl interpreter, with the Ellogon core loaded as a Tcl extension, and a set of Tcl/Tk scripts that implement the {GUI}. The core component of Ellogon, being a Tcl extension, heavily relies on Tcl objects to implement its data model, a decision made more than a decade ago, which poses difficulties into making Ellogon a multi-threaded application.},
	booktitle = {Proceedings of the 17th Annual Tcl/Tk Conference (Tcl 2010)},
	author = {Petasis, Georgios},
	date = {2010-10-11},
}

@inproceedings{petasis_tileqt_2010,
	location = {Hilton Suites Chicago/Oakbrook Terrace, 10 Drury Lane, Oakbrook Terrace, Illinois, United States 60181},
	title = {{TileQt} and {TileGtk}: current status},
	url = {http://www.ellogon.org/petasis/bibliography/Tcl2010/TileQtAndTileGTK.pdf},
	abstract = {This paper is about two Tile and Ttk themes, {TileQt} and {TileGTK}. Despite being two distinct and very different extensions, the motivation for their development was common: making Tk applications look as native as possible under the Linux operating system.},
	booktitle = {Proceedings of the 17th Annual Tcl/Tk Conference (Tcl 2010)},
	author = {Petasis, Georgios},
	date = {2010-10-11},
}

@inproceedings{petasis_tkgecko_2010,
	location = {Hilton Suites Chicago/Oakbrook Terrace, 10 Drury Lane, Oakbrook Terrace, Illinois, United States 60181},
	title = {{TkGecko}: Another Attempt for an {HTML} Renderer for Tk},
	url = {http://www.ellogon.org/petasis/bibliography/Tcl2010/TkGecko.pdf},
	abstract = {The support for displaying {HTML} and especially complex Web sites has always been problematic in Tk. Several efforts have been made in order to alleviate this problem, and this paper presents another (and still incomplete) one. This paper presents {TkGecko}, a Tcl/Tk extension written in C++, which allows Gecko (the {HTML} processing and rendering engine developed by the Mozilla Foundation) to be embedded as a widget in Tk. The current status of the {TkGecko} extension is alpha quality, while the code is publically available under the {BSD} license.},
	booktitle = {Proceedings of the 17th Annual Tcl/Tk Conference (Tcl 2010)},
	author = {Petasis, Georgios},
	date = {2010-10-11},
}

@inproceedings{petasis_tkribbon_2010,
	location = {Hilton Suites Chicago/Oakbrook Terrace, 10 Drury Lane, Oakbrook Terrace, Illinois, United States 60181},
	title = {{TkRibbon}: Windows Ribbons for Tk},
	url = {http://www.ellogon.org/petasis/bibliography/Tcl2010/TkRibbon.pdf},
	abstract = {This paper is about {TkRibbon}, a Tcl/Tk extension that aims to introduce support for the Windows Ribbon Framework in the Tk toolkit. The Windows Ribbon is a graphical interface where a set of toolbars are placed on tabs in a notebook widget, aiming to substitute traditional menus and toolbars. This paper briefly describes Windows Ribbon framework, the {TkRibbon} Tk extension and presents some examples on how {TkRibbon} can be used by Tk applications.},
	booktitle = {Proceedings of the 17th Annual Tcl/Tk Conference (Tcl 2010)},
	author = {Petasis, Georgios},
	date = {2010-10-11},
}

@inproceedings{petasis_blogbuster_2010,
	location = {Valletta, Malta},
	title = {{BlogBuster}: A Tool for Extracting Corpora from the Blogosphere},
	isbn = {2-9517408-6-7},
	url = {http://www.ellogon.org/petasis/bibliography/LREC2010/LREC2010-BlogBuster-CameraReady.pdf},
	abstract = {This paper presents {BlogBuster}, a tool for extracting a corpus from the blogosphere. The topic of cleaning arbitrary web pages with the goal of extracting a corpus from web data, suitable for linguistic and language technology research and development, has attracted significant research interest recently. Several general purpose approaches for removing boilerplate have been presented in the literature; however the blogosphere poses additional requirements, such as a finer control over the extracted textual segments in order to accurately identify important elements, i.e. individual blog posts, titles, posting dates or comments. {BlogBuster} tries to provide such additional details along with boilerplate removal, following a rule-based approach. A small set of rules were manually constructed by observing a limited set of blogs from the Blogger and Wordpress hosting platforms. These rules operate on the {DOM} tree of an {HTML} page, as constructed by a popular browser, Mozilla Firefox. Evaluation results suggest that {BlogBuster} is very accurate when extracting corpora from blogs hosted in the Blogger and Wordpress, while exhibiting a reasonable precision when applied to blogs not hosted in these two popular blogging platforms.},
	booktitle = {Proceedings of the 7th International Conference on Language Resources and Evaluation, {LREC} 2010},
	publisher = {European Language Resources Association},
	author = {Petasis, Georgios and Petasis, Dimitrios},
	editor = {Calzolari, Nicoletta and Choukri, Khalid and Maegaard, Bente and Mariani, Joseph and Odijk, Jan and Piperidis, Stelios and Rosner, Mike and Tapias, Daniel},
	date = {2010-05-17},
}

@inproceedings{petasis_semi-automated_2009,
	location = {Hersonissos, Crete, Greece},
	title = {Semi-automated ontology learning: the {BOEMIE} approach},
	url = {http://www.ellogon.org/petasis/bibliography/ESWC2009/IRMLeS2009-ESWC2009.pdf},
	abstract = {In this paper we describe a semi-automated approach for ontology learning. Exploiting an ontology-based multimodal information extraction system, the ontology learning subsystem accumulates documents that are insufficiently analysed and through clustering proposes new concepts, relations and interpretation rules to be added to the ontology.},
	booktitle = {Proceedings of the First {ESWC} Workshop on Inductive Reasoning and Machine Learning on the Semantic Web ({IRMLeS} 2009), 6th European Semantic Web Conference ({ESWC} 2009)},
	author = {Petasis, Georgios and Karkaletsis, Vangelis and Krithara, Anastasia and Paliouras, Georgios and Spyropoulos, Constantine D.},
	date = {2009-06-01},
	keywords = {evolution, ontologies},
}

@article{castano_multimedia_2009,
	title = {Multimedia Interpretation for Dynamic Ontology Evolution},
	volume = {19},
	url = {http://logcom.oxfordjournals.org/content/19/5/859.abstract},
	doi = {10.1093/logcom/exn049},
	abstract = {The recent success of distributed and dynamic infrastructures for knowledge sharing has raised the need for semiautomatic/automatic ontology evolution strategies. Ontology evolution is generally defined as the timely adaptation of an ontology to changing requirements and the consistent propagation of changes to dependent artifacts. In this article, we present an ontology evolution approach in the context of multimedia interpretation. Ontology evolution in this context relies on the results obtained through reasoning for the interpretation of multimedia resources, through population of the ontology with new individuals or through enrichment of the ontology with new concepts and new semantic relations. The article analyses the results of interpretation, population and enrichment obtained in evaluation experiments in terms of measures such as precision and recall. The evaluation reveals encouraging results.},
	pages = {859--897},
	number = {5},
	journaltitle = {Journal of Logic and Computation},
	author = {Castano, Silvana and Peraldi, Irma Sofia Espinosa and Ferrara, Alfio and Karkaletsis, Vangelis and Kaya, Atila and Möller, Ralf and Montanelli, Stefano and Petasis, Georgios and Wessel, Michael},
	date = {2009},
	note = {\_eprint: http://logcom.oxfordjournals.org/content/19/5/859.full.pdf+html},
}

@incollection{spiliotopoulos_framework_2008,
	location = {Brno, Czech Republic},
	title = {A Framework for Language-Independent Analysis and Prosodic Feature Annotation of Text Corpora},
	volume = {5246},
	isbn = {978-3-540-87390-7},
	url = {http://dx.doi.org/10.1007/978-3-540-87391-4_66},
	series = {Lecture Notes in Computer Science},
	abstract = {Concept-to-Speech systems include Natural Language Generators that produce linguistically enriched text descriptions which can lead to significantly improved quality of speech synthesis. There are cases, however, where either the generator modules produce pieces of non-analyzed, non-annotated plain text, or such modules are not available at all. Moreover, the language analysis is restricted by the usually limited domain coverage of the generator due to its embedded grammar. This work reports on a language-independent framework basis, linguistic resources and language analysis procedures (word/sentence identification, part-of-speech, prosodic feature annotation) for text annotation/processing for plain or enriched text corpora. It aims to produce an automated {XML}- annotated enriched prosodic markup for English and Greek texts, for improved synthetic speech. The markup includes information for both training the synthesizer and for actual input for synthesising. Depending on the domain and target, different methods may be used for automatic classification of entities (words, phrases, sentences) to one or more preset categories such as "emphatic event", "new/old information", "second argument to verb", "proper noun phrase", etc. The prosodic features are classified according to the analysis of the speech-specific characteristics for their role in prosody modelling and passed through to the synthesizer via an extended {SOLE}-{ML} description. Evaluation results show that using selectable hybrid methods for part-of-speech tagging high accuracy is achieved. Annotation of a large generated text corpus containing 50\% enriched text and 50\% canned plain text produces a fully annotated uniform {SOLE}-{ML} output containing all prosodic features found in the initial enriched source. Furthermore, additional automatically-derived prosodic feature annotation and speech synthesis related values are assigned, such as word-placement in sentences and phrases, previous and next word entity relations, emphatic phrases containing proper nouns, and more.},
	pages = {517--524},
	booktitle = {Text, Speech and Dialogue (Proceedings of the 11th International Conference on Text, Speech and Dialogue ({TSD} 2008))},
	publisher = {Springer Berlin / Heidelberg},
	author = {Spiliotopoulos, Dimitris and Petasis, Georgios and Kouroupetroglou, Georgios},
	editor = {Sojka, Petr and Horák, Aleš and Kopeček, Ivan and Pala, Karel},
	date = {2008-09-08},
	doi = {10.1007/978-3-540-87391-4_66},
}

@inproceedings{petasis_segmenting_2008,
	location = {Marrakech, Morocco},
	title = {Segmenting {HTML} pages using visual and semantic information},
	url = {http://www.ellogon.org/petasis/bibliography/LREC2008/LREC-2008-SemanticSegmentation-Submitted.pdf},
	doi = {10.1109/SPCA.2006.297506},
	abstract = {The information explosion of the Web aggravates the problem of effective information retrieval. Even though linguistic approaches found in the literature perform linguistic annotation by creating metadata in the form of tokens, lemmas or part of speech tags, however,this process is insufficient. This is due to the fact that these linguistic metadata do not exploit the actual content of the page, leading to the need of performing semantic annotation based on a predefined semantic model. This paper proposes a new learning approach for performing automatic semantic annotation. This is the result of a two step procedure: the first step partitions a web page into blocks based on its visual layout, while the second, performs subsequent partitioning based on the examination of appearance of specific types of entities denoting the semantic category as well as the application of a number of simple heuristics. Preliminary experiments performed on a manually annotated corpus regarding athletics proved to be very promising.},
	pages = {18--24},
	booktitle = {Proceedings of the 4th Web as a Corpus Workshop ({WAC}-4), 6th Language Resources and Evaluation Conference ({LREC} 2008)},
	author = {Petasis, Georgios and Fragkou, Pavlina and Theodorakos, Aris and Karkaletsis, Vangelis and Spyropoulos, Constantine D.},
	date = {2008-06-01},
}

@inproceedings{fragkou_boemie_2008,
	location = {Marrakech, Morocco},
	title = {{BOEMIE} Ontology-Based Text Annotation Tool},
	url = {http://www.ellogon.org/petasis/bibliography/LREC2008/LREC-2008-324_paper.pdf},
	abstract = {The huge amount of the available information in the Web creates the need of effective information extraction systems that are able to produce metadata that satisfy users information needs. The development of such systems, in the majority of cases, depends on the availability of an appropriately annotated corpus in order to learn extraction models. The production of such corpora can be significantly facilitated by annotation tools that are able to annotate, according to a defined ontology, not only named entities but most importantly relations between them. This paper describes the {BOEMIE} ontology-based annotation tool which is able to locate blocks of text that correspond to specific types of named entities, fill tables corresponding to ontology concepts with those named entities and link the filled tables based on relations defined in the domain ontology. Additionally, it can perform annotation of blocks of text that refer to the same topic. The tool has a user-friendly interface, supports automatic pre-annotation, annotation comparison as well as customization to other annotation schemata. The annotation tool has been used in a large scale annotation task involving 3000 web pages regarding athletics. It has also been used in another annotation task involving 503 web pages with medical information, in different languages.},
	booktitle = {Proceedings of the 6th International Conference on Language Resources and Evaluation ({LREC} 2008)},
	publisher = {European Language Resources Association},
	author = {Fragkou, Pavlina and Petasis, Georgios and Theodorakos, Aris and Karkaletsis, Vangelis and Spyropoulos, Constantine D.},
	date = {2008-06-26},
}

@inproceedings{petasis_learning_2008,
	location = {Amsterdam, The Netherlands, The Netherlands},
	title = {Learning context-free grammars to extract relations from text},
	volume = {178},
	isbn = {978-1-58603-891-5},
	url = {http://www.ellogon.org/petasis/bibliography/ECAI2008/ECAI2008_0371.pdf},
	series = {Frontiers in Artificial Intelligence and Applications},
	abstract = {In this paper we propose a novel relation extraction method, based on grammatical inference. Following a semi-supervised learning approach, the text that connects named entities in an annotated corpus is used to infer a context free grammar. The grammar learning algorithm is able to infer grammars from positive examples only, controlling overgeneralisation through minimum description length. Evaluation results show that the proposed approach performs comparable to the state of the art, while exhibiting a bias towards precision, which is a sign of conservative generalisation.},
	pages = {303--307},
	booktitle = {Proceeding of the 2008 conference on {ECAI} 2008: 18th European Conference on Artificial Intelligence},
	publisher = {{IOS} Press},
	author = {Petasis, Georgios and Karkaletsis, Vangelis and Paliouras, Georgios and Spyropoulos, Constantine D.},
	editor = {Ghallab, Malik and Spyropoulos, Constantine D. and Fakotakis, Nikos and Avouris, Nikolaos M.},
	date = {2008},
}

@inproceedings{castano_ontology_2007,
	location = {Innsbruck, Austria},
	title = {Ontology Dynamics with Multimedia Information: The {BOEMIE} Evolution Methodology},
	url = {http://www.ellogon.org/petasis/bibliography/IWOD2007/IWOD2007-paper-07.pdf},
	abstract = {In this paper, we present the ontology evolution methodology developed in the context of the {BOEMIE} project. Ontology evolution in {BOEMIE} relies on the results obtained through reasoning for the interpretation of multimedia resources in order to evolve (enhance) the ontology, through population of the ontology with new instances, or through enrichment of the ontology with new concepts and new semantic relations.},
	booktitle = {Proceedings of the International {ESWC} Workshop on Ontology Dynamics ({IWOD} 2007)},
	author = {Castano, Silvana and Espinosa, Sofia and Ferrara, Alfio and Karkaletsis, Vangelis and Kaya, Atila and Melzer, Sylvia and Moller, Ralf and Montanelli, Stefano and Petasis, Georgios},
	date = {2007-06-07},
}

@inproceedings{spiliotopoulos_prosodically_2005,
	location = {Patras, Greece},
	title = {Prosodically Enriched Text Annotation for High Quality Speech Synthesis},
	url = {http://www.ellogon.org/petasis/bibliography/SPECOM2005/Spiliotopoulos-SPECOM-2005.pdf},
	abstract = {Linguistically enriched text generated from natural language modules contributes significantly on the quality of speech synthesis. For all cases where such modules are not available, such enriched input needs to be produced from plain text in order to maintain quality. This work reports on a framework of several combined language resources and procedures (word/sentence identification, syntactic analysis, prosodic feature annotation) for text annotation/processing from plain text. Using that, the implementation of an automatic {XML} formatted output generation module produces the prosodically enriched markup.},
	pages = {313--316},
	booktitle = {Proceedings of the 10th International Conference on Speech and Computer ({SPECOM}-2005)},
	author = {Spiliotopoulos, Dimitris and Petasis, Georgios and Kouroupetroglou, Georgios},
	date = {2005-10-17},
}

@inproceedings{petasis_eg-grids_2004,
	location = {Athens, Greece},
	title = {Eg-{GRIDS}: Context-Free Grammatical Inference from Positive Examples Using Genetic Search},
	volume = {3264},
	isbn = {3-540-23410-1},
	url = {http://www.ellogon.org/petasis/bibliography/ICGI2004/e-GRIDS-ICGI-2004-Submission.pdf},
	series = {Lecture Notes in Computer Science},
	abstract = {In this paper we present eg-{GRIDS}, an algorithm for inducing context-free grammars that is able to learn from positive sample sentences. The presented algorithm, similar to its {GRIDS} predecessors, uses simplicity as a criterion for directing inference, and a set of operators for exploring the search space. In addition to the basic beam search strategy of {GRIDS}, eg-{GRIDS} incorporates an evolutionary grammar selection process, aiming to explore a larger part of the search space. Evaluation results are presented on artificially generated data, comparing the performance of beam search and genetic search. These results show that genetic search performs better than beam search while being significantly more efficient computationally.},
	pages = {223--234},
	booktitle = {Grammatical Inference: Algorithms and Applications, Proceedings of the 7th International Colloquium on Grammatical Inference ({ICGI} 2004)},
	publisher = {Springer Berlin / Heidelberg},
	author = {Petasis, Georgios and Paliouras, Georgios and Spyropoulos, Constantine D. and Halatsis, Constantine},
	editor = {Paliouras, Georgios and Sakakibara, Yasubumi},
	date = {2004-10-11},
}

@inproceedings{petasis_adaptive_2004,
	location = {Valencia, Spain},
	title = {Adaptive, Multilingual Named Entity Recognition in Web Pages},
	isbn = {1-58603-452-9},
	url = {http://www.ellogon.org/petasis/bibliography/ECAI2004/Petasis-ECAI2004-Poster.pdf},
	abstract = {Most of the information on the Web today is in the form of {HTML} documents, which are designed for presentation purposes and not for machine understanding and reasoning. Existing web extraction systems require a lot of human involvement for maintenance due to changes to targeted web sites and for adaptation to new web sites or even to new domains. This paper presents the adaptive, multilingual named entity recognition and classification ({NERC}) technologies developed for processing web pages in the context of the R\&D project {CROSSMARC}. The evaluation results demonstrate the viability of our approach.},
	pages = {1073--1074},
	booktitle = {Proceedings of the 16th Eureopean Conference on Artificial Intelligence ({ECAI}'2004), including Prestigious Applicants of Intelligent Systems ({PAIS} 2004)},
	publisher = {{IOS} Press},
	author = {Petasis, Georgios and Karkaletsis, Vangelis and Grover, Claire and Hachey, Ben and Pazienza, Maria Teresa and Vindigni, Michele and Coch, José},
	editor = {Mántaras, Ramon López de and Saitta, Lorenza},
	date = {2004-08-22},
}

@inproceedings{perantonis_text_2004,
	location = {Samos, Greece},
	title = {Text Area Identification in Web Images},
	volume = {3025},
	isbn = {3-540-21937-4},
	url = {http://www.ellogon.org/petasis/bibliography/SETN2004/SETN2004.pdf},
	series = {Lecture Notes in Computer Science},
	abstract = {With the explosive growth of the World Wide Web, millions of documents are published and accessed on-line. Statistics show that a significant part of Web text information is encoded in Web images. Since Web images have special characteristics that sometimes distinguish them from other types of images, commercial {OCR} products often fail to recognize Web images due to their special characteristics. This paper proposes a novel Web image processing algorithm that aims to locate text areas and prepare them for {OCR} procedure with better results. Our methodology for text area identification has been fully integrated with an {OCR} engine and with an Information Extraction system. We present quantitative results for the performance of the {OCR} engine as well as qualitative results concerning its effects to the Information Extraction system. Experimental results obtained from a large corpus of Web images, demonstrate the efficiency of our methodology.},
	pages = {82--92},
	booktitle = {Methods and Applications of Artificial Intelligence, Proceedings of the 3rd Hellenic Conference on Artificial Intelligence ({SETN} 2004)},
	publisher = {Springer Berlin / Heidelberg},
	author = {Perantonis, Stavros J. and Gatos, Basilios and Maragos, Vassilios and Karkaletsis, Vangelis and Petasis, Georgios},
	editor = {Vouros, George A. and Panayiotopoulos, Themis},
	date = {2004-05},
}

@article{petasis_e-grids_2004,
	title = {E-{GRIDS}: Computationally Efficient Grammatical Inference from Positive Examples},
	volume = {7},
	url = {http://www.ellogon.org/petasis/bibliography/GRAMMARS/GRAMMARS2004.pdf},
	abstract = {In this paper we present a new computationally efficient algorithm for inducing context-free grammars that is able to learn from positive sample sentences. This new algorithm uses simplicity as a criterion for directing inference, and the search process of the new algorithm has been optimised by utilising the results of a theoretical analysis regarding the behaviour and complexity of the search operators. Evaluation results are presented on artificially generated data, while the scalability of the algorithm is tested on a large textual corpus. These results show that the new algorithm performs well and can infer grammars from large data sets in a reasonable amount of time.},
	pages = {69--110},
	journaltitle = {{GRAMMARS}},
	author = {Petasis, Georgios and Paliouras, Georgios and Karkaletsis, Vangelis and Halatsis, Constantine and Spyropoulos, Constantine D.},
	date = {2004},
	keywords = {grammatical inference, context-free grammars, minimum description length, positive examples},
}

@inproceedings{petasis_using_2003,
	location = {Thessaloniki, Greece},
	title = {Using the Ellogon Natural Language Engineering Infrastructure},
	url = {http://www.ellogon.org/petasis/bibliography/BCI2003/BCI2003-Petasis.pdf},
	abstract = {Ellogon is a multi-lingual, cross-operating system, general-purpose natural language engineering infrastructure. Ellogon has been used extensively in various {NLP} applications. It is currently provided for free for research use to research and academic organisations. In this paper, we outline its architecture and data model, present Ellogon features as used by different types of users and discuss its functionalities against other infrastructures for language engineering.},
	booktitle = {Proceedings of the Workshop on Balkan Language Resources and Tools, 1st Balkan Conference in Informatics ({BCI} 2003)},
	author = {Petasis, Georgios and Karkaletsis, Vangelis and Paliouras, Georgios and Spyropoulos, Constantine D.},
	date = {2003-11-21},
}

@inproceedings{petasis_cross-lingual_2003,
	location = {Borovets, Bulgaria},
	title = {Cross-lingual Information Extraction from Web pages: the use of a general-purpose Text Engineering Platform},
	url = {http://www.ellogon.org/petasis/bibliography/RANLP2003/RANLP-CameraReady.pdf},
	abstract = {In this paper we present how the use of a general-purpose text engineering platform has facilitated the development of a cross-lingual information extraction system and its adaptation to new domains and languages. Our approach for crosslingual information extraction from the Web covers all the way from the identification of Web sites of interest, to the location of the domain specific Web pages, to the extraction of specific information from the Web pages and its presentation to the end-user. This approach has been implemented in the context of the {IST} project {CROSSMARC}. The text engineering platform "Ellogon" offers functionalities that facilitated the development of core {CROSSMARC} components as well as their porting into new domains and languages.},
	pages = {381--388},
	booktitle = {Proceedings of the 4th International Conference on Recent Advances in Natural Language Processing ({RANLP} 2003)},
	author = {Petasis, Georgios and Karkaletsis, Vangelis and Spyropoulos, Constantine D.},
	date = {2003-09-10},
}

@incollection{petasis_greek_2003,
	title = {A Greek Morphological Lexicon and Its Exploitation by Natural Language Processing Applications},
	volume = {2563},
	isbn = {3-540-07544-5},
	url = {http://www.ellogon.org/petasis/bibliography/PCI2003/25630398.pdf},
	series = {Lecture Notes in Computer Science},
	abstract = {This paper presents a large-scale Greek morphological lexicon, developed at the Software \& Knowledge Engineering Laboratory ({SKEL}) of {NCSR} "Demokritos". The paper describes the lexicon architecture and the procedure to develop and update it. The morphological lexicon was used to develop a lemmatiser and a morphological analyser that were exploited in various natural language processing applications for Greek. The paper presents these applications (controlled language checker, information extraction, information filtering) and discusses further research issues and how we plan to address them.},
	pages = {401--419},
	booktitle = {Advances in Informatics - Post-proceedings of the 8th Panhellenic Conference in Informatics},
	publisher = {Springer Berlin / Heidelberg},
	author = {Petasis, Georgios and Karkaletsis, Vangelis and Farmakiotou, Dimitra and Androutsopoulos, Ion and Spyropoulos, Constantine D.},
	editor = {Manolopoulos, Yannis and Evripidou, Skevos and Kakas, Antonis},
	date = {2003},
	doi = {10.1007/3-540-38076-0_26},
}

@inproceedings{farmakiotou_patedit_2002,
	location = {Las Palmas, Canary Islands, Spain},
	title = {{PatEdit}: An Information Extraction Pattern Editor for Fast System Customization},
	url = {http://www.ellogon.org/petasis/bibliography/LREC2002/LREC2002_Farmakiotou.pdf},
	abstract = {This paper addresses the problem of Information Extraction ({IE}) system customization to new domains and extraction needs with the use of {PatEdit}, an {IE} Pattern Editor. {PatEdit} is a human-assisted knowledge engineering tool, that facilitates the production of {IE} patterns. First, we present the problem of {IE} system customisation and the use of human assisted knowledge engineering tools. Then, we describe {PatEdit} with respect to the {IE} pattern language used and discuss its characteristics that facilitate rapid pattern writing. Finally, the exploitation of {PatEdit} in two information extraction projects is presented along with our plans for future work.},
	pages = {1097--1102},
	booktitle = {Proceedings of the 3rd International Conference on Language Resources and Evaluation ({LREC} 2002)},
	publisher = {European Language Resources Association},
	author = {Farmakiotou, Dimitra and Karkaletsis, Vangelis and Koutsias, Ioannis and Petasis, Georgios and Spyropoulos, Constantine D.},
	date = {2002-05-29},
}

@inproceedings{grover_multilingual_2002,
	location = {Las Palmas, Canary Islands, Spain},
	title = {Multilingual {XML}-Based Named Entity Recognition for E-Retail Domains},
	url = {http://www.ellogon.org/petasis/bibliography/LREC2002/LREC2002_Grover.pdf},
	abstract = {We describe the multilingual Named Entity Recognition and Classification ({NERC}) subpart of an e-retail product comparison system which is currently under development as part of the {EU}-funded project {CROSSMARC}. The system must be rapidly extensible, both to new languages and new domains. To achieve this aim we use {XML} as our common exchange format and the monolingual {NERC} components use a combination of rule-based and machine-learning techniques. It has been challenging to process web pages which contain heavily structured data where text is intermingled with {HTML} and other code. Our preliminary evaluation results demonstrate the viability of our approach.},
	booktitle = {Proceedings of the 3rd International Conference on Language Resources and Evaluation ({LREC} 2002)},
	publisher = {European Language Resources Association},
	author = {Grover, Claire and Mcdonald, Scott and Gearailt, Donnla Nic and Karkaletsis, Vangelis and Farmakiotou, Dimitra and Samaritakis, Georgios and Petasis, Georgios and Pazienza, Maria Teresa and Vindigni, Michele and Vichot, Frantz},
	date = {2002-05-29},
}

@inproceedings{petasis_ellogon_2002,
	location = {Las Palmas, Canary Islands, Spain},
	title = {Ellogon: A New Text Engineering Platform},
	url = {http://www.ellogon.org/petasis/bibliography/LREC2002/LREC2002_Petasis.pdf},
	abstract = {This paper presents Ellogon, a multi-lingual, cross-platform, general-purpose text engineering environment. Ellogon was designed in order to aid both researchers in natural language processing, as well as companies that produce language engineering systems for the end-user. Ellogon provides a powerful {TIPSTER}-based infrastructure for managing, storing and exchanging textual data, embedding and managing text processing components as well as visualising textual data and their associated linguistic information. Among its key features are full Unicode support, an extensive multi-lingual graphical user interface, its modular architecture and the reduced hardware requirements.},
	pages = {72--78},
	booktitle = {Proceedings of the 3rd International Conference on Language Resources and Evaluation ({LREC} 2002)},
	publisher = {European Language Resources Association},
	author = {Petasis, Georgios and Karkaletsis, Vangelis and Paliouras, Georgios and Androutsopoulos, Ion and Spyropoulos, Constantine D.},
	date = {2002-05-29},
}

@inproceedings{farmakiotou_named_2002,
	location = {Thessaloniki, Greece},
	title = {Named Entity Recognition from Greek Web Pages},
	url = {http://www.ellogon.org/petasis/bibliography/SETN2002/091.pdf},
	abstract = {We describe the functionalities of the Hellenic Named Entity Recognition and Classification ({HNERC}) system developed in the context of the {CROSSMARC} project. {CROSSMARC} is developing technology for e-retail product comparison. The {CROSSMARC} system locates relevant retailers web pages and processes them in order to extract information about their products (e.g. technical features, prices). {CROSSMARC}s technology is demonstrated and evaluated for two different product types and four languages (English, Greek, Italian, French). This paper presents the {HNERC} system that is responsible for the identification and classification of specific types of proper names (e.g. laptop manufacturers, models), numerical expressions (e.g. length, weight), and temporal expressions (e.g. time, date) in Hellenic vendor sites. The paper presents the {HNERC} processing stages using examples from the laptops domain.},
	pages = {91--102},
	booktitle = {Proceedings of the 2nd Hellenic Conference on Artificial Intelligence ({SETN}-02), Companion Volume},
	author = {Farmakiotou, Dimitra and Karkaletsis, Vangelis and Samaritakis, Georgios and Petasis, Georgios and Spyropoulos, Constantine D.},
	editor = {Vlahavas, Ioannis P. and Spyropoulos, Constantine D.},
	date = {2002-04-11},
}

@incollection{petasis_symbolic_2002,
	title = {Symbolic and Neural Learning of Named-Entity Recognition and Classification Systems in Two Languages},
	volume = {18},
	isbn = {978-0-7923-7645-3},
	url = {http://www.ellogon.org/petasis/bibliography/COIL2000/COILBook2001.pdf},
	series = {International Series in Intelligent Technologies},
	abstract = {This paper compares two alternative approaches to the problem of acquiring named-entity recognition and classification systems from training corpora, in two different languages. The process of named-entity recognition and classification is an important subtask in most language engineering applications, in particular information extraction, where different types of named entity are associated with specific roles in events. The manual construction of rules for the recognition of named entities is a tedious and time-consuming task. For this reason, effective methods to acquire such systems automatically from data are very desirable. In this paper we compare two popular learning methods on this task: a decision-tree induction method and a multi-layered feed-forward neural network. Particular emphasis is paid on the selection of the appropriate data representation for each method and the extraction of training examples from unstructured textual data. We compare the performance of the two methods on large corpora of English and Greek texts and present the results. In addition to the good performance of both methods, one very interesting result is the fact that a simple representation of the data, which ignores the order of the words within a named entity, leads to improved results over a more complex approach that preserves word order.},
	pages = {193--210},
	booktitle = {Advances in Computational Intelligence and Learning: Methods and Applications},
	publisher = {Springer Berlin / Heidelberg},
	author = {Petasis, Georgios and Petridis, Sergios and Paliouras, Georgios and Karkaletsis, Vangelis and Perantonis, Stavros J. and Spyropoulos, Constantine D.},
	editor = {Zimmermann, Hans-Jurgen and Tselentis, Georgios and Someren, Maarsten van and Dounias, Georgios},
	date = {2002-01},
	keywords = {named entity recognition, neural networks, tree induction},
}

@inproceedings{petasis_greek_2001,
	title = {A Greek Morphological Lexicon and its Exploitation by a Greek Controlled Language Checker},
	url = {http://www.ellogon.org/petasis/bibliography/PCI2001/EPY-Morph-CameraReady.pdf},
	series = {{PCI}'01},
	abstract = {This paper presents a large-scale Greek morphological lexicon, developed by the Software \& Knowledge Engineering Laboratory ({SKEL}) of {NCSR} "Demokritos". The paper describes the lexicon architecture and the procedure to develop and update it. The morphological lexicon was used to develop a lemmatiser and a morphological analyser that were included in a controlled language checker for Greek. The paper discusses the current coverage of the lexicon, as well as remaining issues and how we plan to address them. Our goal is to produce a wide-coverage morphological lexicon of Greek that can be easily exploited in several natural language processing applications.},
	pages = {80--89},
	booktitle = {Proceedings of the 8th Panhellenic Conference on Informatics ({PCI}'01)},
	author = {Petasis, Georgios and Karkaletsis, Vangelis and Farmakiotou, Dimitra and Androutsopoulos, Ion and Spyropoulos, Constantine D.},
	date = {2001-11-08},
}

@inproceedings{petasis_using_2001,
	location = {Toulouse, France},
	title = {Using Machine Learning to Maintain Rule-based Named - Entity Recognition and Classification Systems},
	url = {http://www.ellogon.org/petasis/bibliography/ACL2001/ACL-2001-CameraReady.pdf},
	doi = {http://dx.doi.org/10.3115/1073012.1073067},
	series = {{ACL} '01},
	abstract = {This paper presents a method that assists in maintaining a rule-based named-entity recognition and classification system. The underlying idea is to use a separate system, constructed with the use of machine learning, to monitor the performance of the rule-based system. The training data for the second system is generated with the use of the rule-based system, thus avoiding the need for manual tagging. The disagreement of the two systems acts as a signal for updating the rule-based system. The generality of the approach is illustrated by applying it to large corpora in two different languages: Greek and French. The results are very encouraging, showing that this alternative use of machine learning can assist significantly in the maintenance of rule-based systems.},
	pages = {426--433},
	booktitle = {Proceedings of the 39th Annual Meeting on Association for Computational Linguistics},
	publisher = {Association for Computational Linguistics},
	author = {Petasis, Georgios and Vichot, Frantz and Wolinski, Francis and Paliouras, Georgios and Karkaletsis, Vangelis and Spyropoulos, Constantine D.},
	date = {2001-07-09},
}

@inproceedings{karkaletsis_controlled_2001,
	location = {Pittsburgh, {PA}, {USA}},
	title = {A Controlled Language Checker Based on the Ellogon Text Engineering Platform},
	url = {http://www.ellogon.org/petasis/bibliography/NAACL2001/NAACL01-demo-ABSTRACT.pdf},
	pages = {74--75},
	booktitle = {Proceedings from Language Technologies 2001: The Second Meeting of the North American Chapter of the Association for Computational Linguistics ({NAACL} 2001)},
	publisher = {Carnegie Mellon University},
	author = {Karkaletsis, Vangelis and Samaritakis, Georgios and Petasis, Georgios and Farmakiotou, Dimitra and Androutsopoulos, Ion and Spyropoulos, Constantine D.},
	date = {2001-06-02},
	keywords = {controlled languages, Modern Greek},
}

@inproceedings{paliouras_learning_2000,
	title = {Learning Decision Trees for Named-Entity Recognition and Classification},
	url = {http://www.ellogon.org/petasis/bibliography/ECAI2000/ECAI-2000.pdf},
	series = {{ECAI} 2000},
	abstract = {We propose the use of decision tree induction as a solution to the problem of customising a named-entity recognition and classification ({NERC}) system to a specific domain. A {NERC} system assigns semantic tags to phrases that correspond to named entities, e.g. persons, locations and organisations. Typically, such a system makes use of two language resources: a recognition grammar and a lexicon of known names, classified by the corresponding named-entity types. {NERC} systems have been shown to achieve good results when the domain of application is very specific. However, the construction of the grammar and the lexicon for a new domain is a hard and time-consuming process. We propose the use of decision trees as {NERC} "grammars" and the construction of these trees using machine learning. In order to validate our approach, we tested C4.5 on the identification of person and organisation names involved in management succession events, using data from the sixth Message Understanding Conference. The results of the evaluation are very encouraging showing that the induced tree can outperform a grammar that was constructed manually.},
	booktitle = {Proceedings of the 14th European Conference on Artificial Intelligence ({ECAI} 2000)},
	author = {Paliouras, Georgios and Karkaletsis, Vangelis and Petasis, Georgios and Spyropoulos, Constantine D.},
	date = {2000-08-20},
}

@inproceedings{petasis_machine_2000,
	location = {Chios, Greece},
	title = {Machine Learning and Named-Entity Recognition},
	booktitle = {Proceedings of the 8th {ELSNET} European Summer School on Language and Speech Communication on the subject of Text and Speech Triggered Information Access ({TeSTIA} 2000)},
	author = {Petasis, Georgios},
	date = {2000-07-15},
}

@inproceedings{petasis_automatic_2000,
	location = {New York, {NY}, {USA}},
	title = {Automatic adaptation of proper noun dictionaries through cooperation of machine learning and probabilistic methods},
	isbn = {1-58113-226-3},
	url = {http://www.ellogon.org/petasis/bibliography/SIGIR2000/SIGIR-CameraReady.pdf},
	doi = {http://doi.acm.org/10.1145/345508.345563},
	series = {{SIGIR} '00},
	abstract = {The recognition of Proper Nouns ({PNs}) is considered an important task in the area of Information Retrieval and Extraction. However the high performance of most existing {PN} classifiers heavily depends upon the avail-ability of large dictionaries of domain-specific Proper Nouns, and a certain amount of manual work for rule writing or manual tagging. Though it is not a heavy requirement to rely on some existing {PN} dictionary (of-ten these resources are available on the web), its coverage of a domain corpus may be rather low, in absence of manual updating. In this paper we propose a technique for the automatic updating of a {PN} Dictionary through the cooperation of an inductive and a probabilistic classifier. In our experiments we show that, whenever an existing {PN} Dictionary allows the identification of 50\% of the proper nouns within a corpus, our technique allows, without additional manual effort, the successful recognition of about 90\% of the remaining 50\%.},
	pages = {128--135},
	booktitle = {Proceedings of the 23rd Annual International {ACM} {SIGIR} Conference on Research and Development in Information Retrieval ({SIGIR})},
	publisher = {{ACM}},
	author = {Petasis, Georgios and Cucchiarelli, Alessandro and Velardi, Paola and Paliouras, Georgios and Karkaletsis, Vangelis and Spyropoulos, Constantine D.},
	date = {2000-07-24},
	keywords = {information extraction, machine learning and {IR}, natural language processing for {IR}, text data mining},
}

@inproceedings{petasis_symbolic_2000,
	location = {Chios, Greece},
	title = {Symbolic and Neural Learning for Named-Entity Recognition},
	url = {http://www.ellogon.org/petasis/bibliography/COIL2000/COIL-2000.pdf},
	abstract = {Named-entity recognition involves the identification and classification of named entities in text. This is an important subtask in most language engineering applications, in particular information extraction, where different types of named entity are associated with specific roles in events. The manual construction of rules for the recognition of named entities is a tedious and time-consuming task. For this reason, we present in this paper two approaches to learning named-entity recognition rules from text. The first approach is a decision-tree induction method and the second a multi-layered feed-forward neural network. Particular emphasis is paid on the selection of the appropriate feature set for each method and the extraction of training examples from unstructured textual data. We compare the performance of the two methods on a large corpus of English text and present the results.},
	pages = {58--66},
	booktitle = {Proceedings of European Best Practice Workshops and Symposium on Computational Intelligence and Learning ({COIL} 2000)},
	author = {Petasis, Georgios and Petridis, Sergios and Paliouras, Georgios and Karkaletsis, Vangelis and Perantonis, Stavros J. and Spyropoulos, Constantine D.},
	date = {2000-06-19},
	keywords = {neural networks, tree induction, name entity recognition},
}

@incollection{petasis_using_2000,
	title = {Using Machine Learning Techniques for Part-Of-Speech Tagging in the Greek Language},
	isbn = {978-981-02-4192-6},
	url = {http://www.ellogon.org/petasis/bibliography/HCI1999/EPY99.pdf},
	abstract = {This article investigates the use of Transformation-Based Error-Driven learning for resolving part-of-speech ambiguity in the Greek language. The aim is not only to study the performance, but also to examine its dependence on different thematic domains. Results are presented here for two different test cases: a corpus on "management succession events" and a general-theme corpus. The two experiments show that the performance of this method does not depend on the thematic domain of the corpus, and its accuracy for the Greek language is around 95\%.},
	pages = {273--281},
	booktitle = {{ADVANCES} {IN} {INFORMATICS}: Proceedings of the 7th Hellenic Conference on Informatics ({HCI} '99)},
	publisher = {World Scientific},
	author = {Petasis, Georgios and Paliouras, Georgios and Karkaletsis, Vangelis and Spyropoulos, Constantine D. and Androutsopoulos, Ion},
	editor = {Fotiadis, Dimitrios I. and Nikolopoulos, Stavros D.},
	date = {2000-05},
}

@article{karkaletsis_named-entity_1999,
	title = {Named-Entity Recognition from Greek and English Texts},
	volume = {26},
	issn = {0921-0296},
	url = {http://www.ellogon.org/petasis/bibliography/JIRS1999/JIRS-1999.pdf},
	doi = {10.1023/A:1008124406923},
	abstract = {Named-entity recognition ({NER}) involves the identification and classification of named entities in text. This is an important subtask in most language engineering applications, in particular information extraction, where different types of named entity are associated with specific roles in events. In this paper, we present a prototype {NER} system for Greek texts that we developed based on a {NER} system for English. Both systems are evaluated on corpora of the same domain and of similar size. The time-consuming process for the construction and update of domain-specific resources in both systems led us to examine a machine learning method for the automatic construction of such resources for a particular application in a specific language.},
	pages = {123--135},
	number = {2},
	journaltitle = {Journal of Intelligent and Robotic Systems},
	author = {Karkaletsis, Vangelis and Paliouras, Georgios and Petasis, Georgios and Manousopoulou, Natasa and Spyropoulos, Constantine D.},
	date = {1999-10},
	note = {Place: Hingham, {MA}, {USA}
Publisher: Kluwer Academic Publishers},
	keywords = {information extraction, machine learning, named-entity recognition},
}

@inproceedings{petasis_exploiting_1999,
	location = {Chania, Greece},
	title = {Exploiting Learning in Bilingual Named Entity Recognition},
	url = {http://www.ellogon.org/petasis/bibliography/ACAI1999/ss1_07.pdf},
	booktitle = {Proceedings of the {ECCAI} Advanced Course on Artificial Intelligence ({ACAI} '99)},
	author = {Petasis, Georgios},
	date = {1999-07-05},
}

@inproceedings{petasis_resolving_1999,
	location = {Chania, Greece},
	title = {Resolving Part-of-Speech Ambiguity in the Greek Language Using Learning Techniques},
	url = {http://www.ellogon.org/petasis/bibliography/ACAI1999/9906019.pdf},
	abstract = {This article investigates the use of Transformation-Based Error-Driven learning for resolving part-of-speech ambiguity in the Greek language. The aim is not only to study the performance, but also to examine its dependence on different thematic domains. Results are presented here for two different test cases: a corpus on "management succession events" and a general-theme corpus. The two experiments show that the performance of this method does not depend on the thematic domain of the corpus, and its accuracy for the Greek language is around 95\%.},
	booktitle = {Proceedings of the {ECCAI} Advanced Course on Artificial Intelligence ({ACAI} '99)},
	author = {Petasis, Georgios and Paliouras, Georgios and Karkaletsis, Vangelis and Spyropoulos, Constantine D.},
	date = {1999-07-05},
}

@incollection{karkaletsis_named_1999,
	title = {Named Entity Recognition from Greek texts: the {GIE} Project},
	volume = {21},
	isbn = {978-1-4020-0393-6},
	url = {http://www.springer.com/computer/image+processing/book/978-1-4020-0393-6},
	series = {Intelligent Systems, Control and Automation: Science and Engineering},
	pages = {131--142},
	booktitle = {Advances in Intelligent Systems: Concepts, Tools and Applications},
	publisher = {Springer Berlin / Heidelberg},
	author = {Karkaletsis, Vangelis and Spyropoulos, Constantine D. and Petasis, Georgios},
	editor = {Tzafestas, Spyros G.},
	date = {1999},
	note = {Section: 12},
	keywords = {information extraction, machine learning, named-entity recognition},
}
