/* https://github.com/FlamingTempura/bibtex-tidy
/* https://flamingtempura.github.io/bibtex-tidy/index.html?opt=%7B%22curly%22%3Atrue%2C%22numeric%22%3Atrue%2C%22space%22%3A2%2C%22tab%22%3Atrue%2C%22align%22%3A13%2C%22duplicates%22%3A%5B%22key%22%5D%2C%22stripEnclosingBraces%22%3Afalse%2C%22dropAllCaps%22%3Afalse%2C%22escape%22%3Afalse%2C%22sortFields%22%3A%5B%22title%22%2C%22shorttitle%22%2C%22author%22%2C%22year%22%2C%22month%22%2C%22day%22%2C%22journal%22%2C%22booktitle%22%2C%22location%22%2C%22on%22%2C%22publisher%22%2C%22address%22%2C%22series%22%2C%22volume%22%2C%22number%22%2C%22pages%22%2C%22doi%22%2C%22isbn%22%2C%22issn%22%2C%22url%22%2C%22urldate%22%2C%22copyright%22%2C%22category%22%2C%22note%22%2C%22metadata%22%5D%2C%22stripComments%22%3Afalse%2C%22trailingCommas%22%3Afalse%2C%22encodeUrls%22%3Afalse%2C%22tidyComments%22%3Atrue%2C%22removeEmptyFields%22%3Afalse%2C%22removeDuplicateFields%22%3Afalse%2C%22lowercase%22%3Atrue%2C%22backup%22%3Atrue%7D

@article{DBLP:journals/aim/KoubarakisVCPTK18,
    title        = {{AI} in Greece: The Case of Research on Linked Geospatial Data},
    author       = {Manolis Koubarakis and George A. Vouros and Georgios Chalkiadakis and Vassilis P. Plagianakos and Christos Tjortjis and Ergina Kavallieratou and Dimitris Vrakas and Nikolaos Mavridis and Georgios Petasis and Konstantinos Blekas and Anastasia Krithara},
    year         = 2018,
    journal      = {{AI} Magazine},
    volume       = 39,
    number       = 2,
    pages        = {91--96},
    url          = {https://www.aaai.org/ojs/index.php/aimagazine/article/view/2801},
    timestamp    = {Tue, 17 Jul 2018 01:00:00 +0200},
    biburl       = {https://dblp.org/rec/bib/journals/aim/KoubarakisVCPTK18},
    bibsource    = {dblp computer science bibliography, https://dblp.org}
}
@inproceedings{Papachristopoulos-QQML:2018,
    title        = {Introducing Sentiment Analysis for the Evaluation of Library's Services Effectiveness},
    author       = {Leonidas Papachristopoulos and Pantelis Ampatzoglou and Ioanna Seferli and Andriani Zafeiropoulou and Georgios Petasis},
    year         = 2018,
    month        = {May},
    day          = {22--25},
    booktitle    = {Proceedings of the 10th Qualitative and Quantitative Methods in Libraries International Conference (QQML2018)},
    address      = {Chania, Greece}
}
@inproceedings{Petasis-DataEconomy:2017,
    title        = {YourDataStories: Transparency and Corruption Fighting through Data Interlinking and Visual Exploration},
    author       = {Georgios Petasis and Anna Triantafillou and Eric Karstens},
    year         = 2017,
    month        = {November},
    day          = 22,
    booktitle    = {Proceedings of the Data Economy Workshop, 4th International Conference on Internet Science (INSCI 2017)},
    address      = {Thessaloniki, Greece}
}
@inproceedings{ferrara-montanelli-petasis:2017:ArgumentMining,
    title        = {Unsupervised Detection of Argumentative Units though Topic Modeling Techniques},
    author       = {Ferrara, Alfio  and  Montanelli, Stefano  and  Petasis, Georgios},
    year         = 2017,
    month        = {September},
    booktitle    = {Proceedings of the 4th Workshop on Argument Mining, 2017 Conference on Empirical Methods in Natural Language Processing (EMNLP 2017)},
    publisher    = {Association for Computational Linguistics},
    address      = {Copenhagen, Denmark},
    pages        = {97--107},
    url          = {http://www.aclweb.org/anthology/W17-5113},
    abstract     = {In this paper we present a new unsupervised approach, "Attraction to Topics" -- A2T , for the detection of argumentative units, a sub-task of argument mining. Motivated by the importance of topic identification in manual annotation, we examine whether topic modeling can be used for performing unsupervised detection of argumentative sentences, and to what extend topic modeling can be used to classify sentences as claims and premises. Preliminary evaluation results suggest that topic information can be successfully used for the detection of argumentative sentences, at least for corpora used for evaluation. Our approach has been evaluated on two English corpora, the first of which contains 90 persuasive essays, while the second is a collection of 340 documents from user generated content.}
}
@inproceedings{Petasis-EtAl:2016:ARG-MINING,
    title        = {Identifying Argument Components through TextRank},
    author       = {Georgios Petasis and Vangelis Karkaletsis},
    year         = 2016,
    month        = {August},
    booktitle    = {Proceedings of the 3rd Workshop on Argument Mining (ArgMining2016), 54th Annual Meeting of the Association for Computational Linguistics (ACL 2016)},
    publisher    = {Association for Computational Linguistics},
    address      = {Berlin, Germany},
    pages        = {56--66},
    url          = {http://argmining2016.arg.tech/},
    url          = {http://aclweb.org/anthology/W/W16/#2800},
    url          = {http://aclweb.org/anthology/W/W16/W16-2811.pdf}
}
@inproceedings{DBLP:conf/lrec/KatakisPK16,
    title        = {{CLARIN-EL} Web-based Annotation Tool},
    author       = {Ioannis Manousos Katakis and Georgios Petasis and Vangelis Karkaletsis},
    year         = 2016,
    booktitle    = {Proceedings of the Tenth International Conference on Language Resources and Evaluation {LREC} 2016, Portoro{\v{z}}, Slovenia, May 23-28, 2016.},
    publisher    = {European Language Resources Association {(ELRA)}},
    url          = {http://www.lrec-conf.org/lrec2016},
    url          = {http://www.lrec-conf.org/proceedings/lrec2016/summaries/990.html},
    editor       = {Nicoletta Calzolari and Khalid Choukri and Thierry Declerck and Sara Goggi and Marko Grobelnik and Bente Maegaard and Joseph Mariani and H{\'{e}}l{\`{e}}ne Mazo and Asunci{\'{o}}n Moreno and Jan Odijk and Stelios Piperidis},
    timestamp    = {Tue, 30 Aug 2016 18:49:47 +0200},
    biburl       = {http://dblp.uni-trier.de/rec/bib/conf/lrec/KatakisPK16},
    bibsource    = {dblp computer science bibliography, http://dblp.org}
}
@article{doi:10.1142/S0218213015400242,
    title        = {Argument Extraction from News, Blogs, and the Social Web},
    author       = {Goudas, Theodosis and Louizos, Christos and Petasis, Georgios and Karkaletsis, Vangelis},
    year         = 2015,
    journal      = {International Journal on Artificial Intelligence Tools},
    volume       = 24,
    number       = {05},
    pages        = 1540024,
    doi          = {10.1142/S0218213015400242},
    url          = {http://www.worldscientific.com/doi/abs/10.1142/S0218213015400242},
    eprint       = {http://www.worldscientific.com/doi/pdf/10.1142/S0218213015400242}
}
@inproceedings{ref35,
    title        = {Predicting Sentiment using Tranfer Learning},
    author       = {Anastasia Krithara and George Giannakopoulos and George Paliouras and George Petasis and Vangelis Karkaletsis},
    year         = 2015,
    booktitle    = {Workshop on Replicability and Reproducibility in Natural Language Processing: adaptive methods, resources and software at IJCAI 2015 (AdaptiveNLP 2015)},
    abstract     = {A new transfer learning method is presented in this paper, addressing the task of sentiment analysis across domains.The proposed approach is a transfer variant of the Probabilistic Latent Semantic Analysis (PLSA) model that we name KLIEP-PLSA. The approach captures the difference of the tributions between the different domains. We perform experiments over well known datasets and show the promising results that we obtained new method.},
    keywords     = {transfer learning,KLIEP-PLSA,PLSA,sentiment analysis}
}
@inproceedings{sardianos-EtAl:2015:ARG-MINING,
    title        = {Argument Extraction from News},
    author       = {Sardianos, Christos  and  Katakis, Ioannis Manousos  and  Petasis, Georgios  and  Karkaletsis, Vangelis},
    year         = 2015,
    month        = {June},
    booktitle    = {Proceedings of the 2nd Workshop on Argumentation Mining},
    publisher    = {Association for Computational Linguistics},
    address      = {Denver, CO},
    pages        = {56--66},
    url          = {http://www.aclweb.org/anthology/W15-0508}
}
@inproceedings{DBLP:conf/lrec/Petasis14,
    title        = {Annotating Arguments: The NOMAD Collaborative Annotation Tool},
    author       = {Georgios Petasis},
    year         = 2014,
    booktitle    = {LREC},
    booktitle    = {Proceedings of the Ninth International Conference on Language Resources and Evaluation (LREC-2014), Reykjavik, Iceland, May 26-31, 2014},
    publisher    = {European Language Resources Association (ELRA)},
    pages        = {1930--1937},
    ee           = {http://www.lrec-conf.org/proceedings/lrec2014/summaries/669.html},
    editor       = {Nicoletta Calzolari and Khalid Choukri and Thierry Declerck and Hrafn Loftsson and Bente Maegaard and Joseph Mariani and Asunci{\'o}n Moreno and Jan Odijk and Stelios Piperidis},
    bibsource    = {DBLP, http://dblp.uni-trier.de}
}
@inproceedings{DBLP:conf/lrec/Petasis14a,
    title        = {The Ellogon Pattern Engine: Context-free Grammars over Annotations},
    author       = {Georgios Petasis},
    year         = 2014,
    booktitle    = {Proceedings of the Ninth International Conference on Language Resources and Evaluation (LREC-2014), Reykjavik, Iceland, May 26-31, 2014},
    publisher    = {European Language Resources Association (ELRA)},
    pages        = {2460--2465},
    ee           = {http://www.lrec-conf.org/proceedings/lrec2014/summaries/1060.html},
    editor       = {Nicoletta Calzolari and Khalid Choukri and Thierry Declerck and Hrafn Loftsson and Bente Maegaard and Joseph Mariani and Asunci{\'o}n Moreno and Jan Odijk and Stelios Piperidis},
    bibsource    = {DBLP, http://dblp.uni-trier.de}
}
@inproceedings{DBLP:conf/lrec/KiomourtzisGPKK14,
    title        = {NOMAD: Linguistic Resources and Tools Aimed at Policy Formulation and Validation},
    author       = {George Kiomourtzis and George Giannakopoulos and Georgios Petasis and Pythagoras Karampiperis and Vangelis Karkaletsis},
    year         = 2014,
    booktitle    = {Proceedings of the Ninth International Conference on Language Resources and Evaluation (LREC-2014), Reykjavik, Iceland, May 26-31, 2014},
    publisher    = {European Language Resources Association (ELRA)},
    pages        = {3464--3470},
    ee           = {http://www.lrec-conf.org/proceedings/lrec2014/summaries/813.html},
    editor       = {Nicoletta Calzolari and Khalid Choukri and Thierry Declerck and Hrafn Loftsson and Bente Maegaard and Joseph Mariani and Asunci{\'o}n Moreno and Jan Odijk and Stelios Piperidis},
    bibsource    = {DBLP, http://dblp.uni-trier.de}
}
@inbook{Goudas2014,
    title        = {Argument Extraction from News, Blogs, and Social Media},
    author       = {Goudas, Theodosis and Louizos, Christos and Petasis, Georgios and Karkaletsis, Vangelis},
    year         = 2014,
    booktitle    = {Artificial Intelligence: Methods and Applications: 8th Hellenic Conference on AI, SETN 2014, Ioannina, Greece, May 15-17, 2014. Proceedings},
    publisher    = {Springer International Publishing},
    address      = {Cham},
    pages        = {287--299},
    doi          = {10.1007/978-3-319-07064-3_23},
    isbn         = {978-3-319-07064-3},
    url          = {http://dx.doi.org/10.1007/978-3-319-07064-3_23},
    editor       = {Likas, Aristidis and Blekas, Konstantinos and Kalles, Dimitris}
}
@inproceedings{SETN-2014-Petasis,
    title        = {Sentiment Analysis for Reputation Management: Mining the Greek Web},
    author       = {Georgios Petasis and Dimitris Spiliotopoulos and Nikos Tsirakis and Panayotis Tsantilas},
    year         = 2014,
    booktitle    = {Artificial Intelligence: Methods and Applications - 8th Hellenic Conference on AI, SETN 2014, Ioannina, Greece, May 15-17, 2014. Proceedings},
    publisher    = {Springer},
    series       = {Lecture Notes in Computer Science},
    volume       = 8445,
    pages        = {327--340},
    isbn         = {978-3-319-07063-6, 978-3-319-07064-3},
    editor       = {Aristidis Likas and Konstantinos Blekas and Dimitris Kalles},
    ee           = {http://dx.doi.org/10.1007/978-3-319-07064-3_26},
    bibsource    = {DBLP, http://dblp.uni-trier.de}
}
@inproceedings{pathos-2013-Petasis,
    title        = {Large-scale Sentiment Analysis for Reputation Management},
    author       = {Georgios Petasis and Dimitrios Spiliotopoulos and Nikos Tsirakis and Panayiotis Tsantilas},
    year         = 2013,
    month        = {September 23},
    booktitle    = {Proceedings of the 2nd Workshop on Practice and Theory of Opinion Mining and Sentiment Analysis (PATHOS-2013)},
    address      = {Darmstadt, Germany},
    abstract     = {Harvesting the web and social web data is a meticulous and complex task. Applying the results to a successful business case such as brand monitoring requires high precision and recall for the opinion mining and entity recognition tasks. This work reports on the integrated platform of a state of the art Named-entity Recognition and Classification (NERC) system and opinion mining methods for a Software-as-a-Service (SaaS) approach on a fully automatic service for brand monitoring for the Greek language. The service has been successfully deployed to the biggest search engine in Greece powering the large-scale linguistic and sentiment analysis of about 80.000 resources per hour.},
    editor       = {Stefan Gindl and Robert Remus and Michael Wiegand}
}
@inproceedings{DBLP:conf/otm/Petasis13,
    title        = {Structuring the Blogosphere on News from Traditional Media},
    author       = {Georgios Petasis},
    year         = 2013,
    month        = {September 9--13},
    booktitle    = {On the Move to Meaningful Internet Systems: OTM 2013 Workshops - Confederated International Workshops: OTM Academy, OTM Industry Case Studies Program, ACM, EI2N, ISDE, META4eS, ORM, SeDeS, SINCOM, SMS, and SOMOCO 2013},
    address      = {Graz, Austria},
    pages        = {608--617},
    ee           = {http://dx.doi.org/10.1007/978-3-642-41033-8_77},
    bibsource    = {DBLP, http://dblp.uni-trier.de}
}
@inproceedings{DBLP:conf/lpnmr/PetasisMK13,
    title        = {BOEMIE: Reasoning-based Information Extraction},
    author       = {Georgios Petasis and Ralf M{\"o}ller and Vangelis Karkaletsis},
    year         = 2013,
    month        = {September 15},
    address      = {A Corunna, Spain},
    pages        = {60--75},
    abstract     = {This paper presents a novel approach for exploiting an ontology in an ontology-based information extraction system, which substitutes part of the extraction process with reasoning, guided by a set of automatically acquired rules.},
    ee           = {http://ceur-ws.org/Vol-1044/paper-06.pdf},
    bibsource    = {DBLP, http://dblp.uni-trier.de}
}
@incollection{tsd-2012-Petasis-Tsoumari,
    title        = {{A} {N}ew {A}nnotation {T}ool for {A}ligned {B}ilingual {C}orpora},
    author       = {Petasis, Georgios and Tsoumari, Mara},
    year         = 2012,
    month        = {September 3--7},
    booktitle    = {Text, Speech and Dialogue},
    publisher    = {Springer Berlin Heidelberg},
    address      = {Brno, Czech Republic},
    series       = {Lecture Notes in Computer Science},
    volume       = 7499,
    pages        = {95--104},
    doi          = {10.1007/978-3-642-32790-2_11},
    isbn         = {978-3-642-32789-6},
    url          = {http://www.tsdconference.org/tsd2012/abstracts.html#I450},
    url          = {http://dx.doi.org/10.1007/978-3-642-32790-2_11},
    url          = {http://www.ellogon.org/petasis/bibliography/TSD2012/tsd450.pdf},
    abstract     = {This paper presents a new annotation tool for aligned bilingual corpora, which allows the annotation of a wide range of information, ranging from information about words (such as part-of-speech tags or named-entities) to quite complex annotation schemas involving links between aligned segments, such as co-reference or translation equivalence between aligned segments in the two languages. The annotation tool is implemented as a component of the Ellogon language engineering platform, exploiting its extensive annotation engine, its cross-platform abilities and its linguistic processing components, if such a need arises. The new annotation tool is distributed with an open source license (LGPL), as part of the Ellogon language engineering platform.},
    booksubtitle = {15th International Conference, TSD 2012, Brno, Czech Republic, September 3--7, 2012. Proceedings},
    keywords     = {Annotation tools, collaborative annotation, adaptable annotation schemas},
    editor       = {Sojka, Petr and Hor\'{a}k, Ale\v{s} and Kope\v{c}ek, Ivan and Pala, Karel},
    keywords     = {Annotation tools; collaborative annotation; adaptable annotation schemas}
}
@inproceedings{lrec-2012-Petasis,
    title        = {{T}he {SYNC}3 {C}ollaborative {A}nnotation {T}ool},
    author       = {Georgios Petasis},
    year         = 2012,
    month        = {May},
    booktitle    = {Proceedings of the 8th International Conference on Language Resources and Evaluation, LREC 2012},
    publisher    = {European Language Resources Association},
    address      = {Istanbul, Turkey},
    pages        = {363--370},
    url          = {http://www.ellogon.org/petasis/bibliography/LREC2012/LREC2012-700.pdf},
    abstract     = {The huge amount of the available information in the Web creates the need for effective information extraction systems that are able to produce metadata that satisfy user's information needs. The development of such systems, in the majority of cases, depends on the availability of an appropriately annotated corpus in order to learn or evaluate extraction models. The production of such corpora can be significantly facilitated by annotation tools, which provide user-friendly facilities and enable annotators to annotate documents according to a predefined annotation schema. However, the construction of annotation tools that operate in a distributed environment is a challenging task: the majority of these tools are implemented as Web applications, having to cope with the capabilities offered by browsers. This paper describes the SYNC3 collaborative annotation tool, which implements an alternative architecture: it remains a desktop application, fully exploiting the advantages of desktop applications, but provides collaborative annotation through the use of a centralised server for storing both the documents and their metadata, and instance messaging protocols for communicating events among all annotators. The annotation tool is implemented as a component of the Ellogon language engineering platform, exploiting its extensive annotation engine, its cross-platform abilities and its linguistic processing components, if such a need arises. Finally, the SYNC3 annotation tool is distributed with an open source license, as part of the Ellogon platform.},
    keywords     = {annotation tools, collaborative annotation, adaptable annotation schemas}
}
@incollection{Iosif.IGIGLOBAL.2012,
    title        = {{O}ntology-{B}ased {I}nformation {E}xtraction under a {B}ootstrapping {A}pproach},
    author       = {Elias Iosif and Georgios Petasis and Vangelis Karkaletsis},
    year         = 2012,
    month        = {April},
    booktitle    = {Semi-Automatic Ontology Development: Processes and Resources},
    publisher    = {IGI Global},
    address      = {Hershey, PA, USA},
    pages        = {1--21},
    doi          = {10.4018/978-1-4666-0188-8.ch001},
    isbn         = 9781466601888,
    url          = {http://www.igi-global.com/chapter/ontology-based-information-extraction-under/63896},
    abstract     = {The authors present an ontology-based information extraction process, which operates in a bootstrapping framework. The novelty of this approach lies in the continuous semantics extraction from textual content in order to evolve the underlying ontology, while the evolved ontology enhances in turn the information extraction mechanism. This process was implemented in the context of the R{\&}D project BOEMIE. The BOEMIE system was evaluated on the athletics domain.},
    editor       = {Maria Teresa Pazienza, Armando Stellato},
    chapter      = 1
}
@inproceedings{eChallenges-2011-Sarris,
    title        = {{A} {S}ystem for {S}ynergistically {S}tructuring {N}ews {C}ontent from {T}raditional {M}edia and the {B}logosphere},
    author       = {Nikos Sarris and Gerasimos Potamianos and Jean-Michel Renders and Claire Grover and Eric Karstens and Leonidas Kallipolitis and Vasilis Tountopoulos and Petasis, Georgios and Anastasia Krithara and Matthias Gall{\'e} and Guillaume Jacquet and Beatrice Alex and Richard Tobin and Liliana Bounegru},
    year         = 2011,
    month        = {October 26--28},
    booktitle    = {eChallenges e-2011 Conference Proceedings},
    address      = {Florence, Italy},
    isbn         = {978-1-905824-27-4},
    url          = {http://www.ellogon.org/petasis/bibliography/eChallenges2011/echallenges_ref_81_doc_7322.pdf},
    abstract     = {News and social media are emerging as a dominant source of information for numerous applications. However, their vast unstructured content present challenges to efficient extraction of such information. In this paper, we present the SYNC3 system that aims to intelligently structure content from both traditional news media and the blogosphere. To achieve this goal, SYNC3 incorporates innovative algorithms that first model news media content statistically, based on fine clustering of articles into so-called {"}news events{"}. Such models are then adapted and applied to the blogosphere domain, allowing its content to map to the traditional news domain. Furthermore, appropriate algorithms are employed to extract news event labels and relations between events, in order to efficiently present news content to the system end users.},
    editor       = {Paul Cunningham and Miriam Cunningham},
    organization = {IIMC International Information Management Corporation}
}
@inproceedings{AEPC2-RANLP-2011-Tsoumari,
    title        = {{C}oreference {A}nnotator - {A} new annotation tool for aligned bilingual corpora},
    author       = {Tsoumari, Mara and Petasis, Georgios},
    year         = 2011,
    month        = {September 15},
    booktitle    = {Proceedings of the Second Workshop on Annotation and Exploitation of Parallel Corpora (AEPC 2), in 8th International Conference on Recent Advances in Natural Language Processing (RANLP 2011)},
    pages        = {43--52},
    url          = {http://www.aclweb.org/anthology/W11-4307},
    abstract     = {This paper presents the main features of an annotation tool, the Coreference Annotator, which manages bilingual corpora consisting of aligned texts that can be grouped in collections and subcollections according to their topics and discourse. The tool allows the manual annotation of certain linguistic items in the source text and their translation equivalent in the target text, by entering useful information about these items based on their context.}
}
@inproceedings{petasis:2011:RANLP,
    title        = {Unsupervised Domain Adaptation based on Text Relatedness},
    author       = {Petasis, Georgios},
    year         = 2011,
    month        = {September 12--14},
    booktitle    = {Proceedings of the International Conference Recent Advances in Natural Language Processing 2011},
    publisher    = {RANLP 2011 Organising Committee},
    address      = {Hissar, Bulgaria},
    pages        = {733--739},
    url          = {http://www.ellogon.org/petasis/bibliography/RANLP2011/ranlp2011-25-Petasis-CameraReady.pdf},
    url          = {http://aclweb.org/anthology/R11-1107},
    abstract     = {In this paper an unsupervised approach to do-main adaptation is presented, which exploits external knowledge sources in order to port a classification model into a new thematic do-main. Our approach extracts a new feature set from documents of the target domain, and tries to align the new features to the original ones, by exploiting text relatedness from external knowledge sources, such as WordNet. The approach has been evaluated on the task of document classification, involving the classification of newsgroup postings into 20 news groups.}
}
@phdthesis{PhD-2011-Petasis,
    title        = {{M}achine {L}earning in {N}atural {L}anguage {P}rocessing},
    author       = {Petasis, Georgios},
    year         = 2011,
    month        = {July 1},
    url          = {http://www.ellogon.org/petasis/bibliography/Petasis/Ph.D.Thesis-GeorgiosPetasis.pdf},
    abstract     = {This thesis examines the use of machine learning techniques in various tasks of natural language processing, mainly for the task of information extraction from texts. The objectives are the improvement of adaptability of information extraction systems to new thematic domains (or even languages), and the improvement of their performance using as fewer resources (either linguistic or human) as possible. This thesis has examined two main axes: a) the research and assessment of existing algorithms of machine learning mainly in the stages of linguistic pre-processing (such as part of speech tagging) and named-entity recognition, and b) the creation of a new machine learning algorithm and its assessment on synthetic data, as well as in real world data from the task of relation extraction between named entities. This new algorithm belongs to the category of inductive grammar learning, and can infer context free grammars from positive examples only.},
    keywords     = {information extraction, machine learning, grammatical inference},
    school       = {Department of Informatics and Telecommunications, University of Athens},
    type         = {Ph.D. Thesis}
}
@incollection{springerlink:10.1007/978-3-642-20795-2_4,
    title        = {{O}ntology {B}ased {I}nformation {E}xtraction from {T}ext},
    author       = {Vangelis Karkaletsis and Pavlina Fragkou and Petasis, Georgios and Elias Iosif},
    year         = 2011,
    booktitle    = {Knowledge-Driven Multimedia Information Extraction and Ontology Evolution},
    publisher    = {Springer Berlin / Heidelberg},
    series       = {Lecture Notes in Computer Science},
    volume       = 6050,
    pages        = {89--109},
    doi          = {10.1007/978-3-642-20795-2_4},
    isbn         = {978-3-642-20794-5},
    url          = {http://dx.doi.org/10.1007/978-3-642-20795-2_4},
    note         = {10.1007/978-3-642-20795-2_4},
    editor       = {Georgios Paliouras and Constantine D. Spyropoulos and George Tsatsaronis},
    abstract     = {Information extraction systems employ ontologies as a means to describe formally the domain knowledge exploited by these systems for their operation. The aim of this survey is to study the contribution of ontologies to information extraction systems. We believe that this will help towards specifying a concrete methodology for ontology based information extraction exploiting all levels of ontological knowledge, from domain entities for named entity recognition, to the use of conceptual hierarchies for pattern generalization, to the use of properties and non-taxonomic relations for pattern acquisition, and finally to the use of the domain model itself for integrating extracted entities and instances of relations, as well as for discovering implicit information and detecting inconsistencies.}
}
@incollection{springerlink:10.1007/978-3-642-20795-2_6,
    title        = {{O}ntology {P}opulation and {E}nrichment: {S}tate of the {A}rt},
    author       = {Petasis, Georgios and Vangelis Karkaletsis and Georgios Paliouras and Anastasia Krithara and Elias Zavitsanos},
    year         = 2011,
    booktitle    = {Knowledge-Driven Multimedia Information Extraction and Ontology Evolution},
    publisher    = {Springer Berlin / Heidelberg},
    series       = {Lecture Notes in Computer Science},
    volume       = 6050,
    pages        = {134--166},
    doi          = {10.1007/978-3-642-20795-2_4},
    isbn         = {978-3-642-20794-5},
    url          = {http://dx.doi.org/10.1007/978-3-642-20795-2_6},
    note         = {10.1007/978-3-642-20795-2_6},
    abstract     = {Ontology learning is the process of acquiring (constructing or integrating) an ontology (semi-) automatically. Being a knowledge acquisition task, it is a complex activity, which becomes even more complex in the context of the BOEMIE project, due to the management of multimedia resources and the multi-modal semantic interpretation that they require. The purpose of this chapter is to present a survey of the most relevant methods, techniques and tools used for the task of ontology learning. Adopting a practical perspective, an overview of the main activities involved in ontology learning is presented. This breakdown of the learning process is used as a basis for the comparative analysis of existing tools and approaches. The comparison is done along dimensions that emphasize the particular interests of the BOEMIE project. In this context, ontology learning in BOEMIE is treated and compared to the state of the art, explaining how BOEMIE addresses problems observed in existing systems and contributes to issues that are not frequently considered by existing approaches.},
    editor       = {Georgios Paliouras and Constantine D. Spyropoulos and George Tsatsaronis}
}
@inproceedings{Tcl-2010-TkDND,
    title        = {{T}k{DND}: a cross-platform drag'n'drop package},
    author       = {Petasis, Georgios},
    year         = 2010,
    month        = {October 11--15},
    booktitle    = {Proceedings of the 17th Annual Tcl/Tk Conference (Tcl 2010)},
    address      = {Hilton Suites Chicago/Oakbrook Terrace, 10 Drury Lane, Oakbrook Terrace, Illinois, United States 60181},
    url          = {http://www.ellogon.org/petasis/bibliography/Tcl2010/TkDND.pdf},
    abstract     = {This paper is about TkDND, a Tcl/Tk extension that aims to add cross-application drag and drop support to Tk, for popular operating systems, such as Microsoft Windows, Apple OS X and GNU/Linux. Being in its second rewrite, TkDND 2.x has a stable implementation for Windows and OS X, while support for Linux and the XDND protocol is still under development.}
}
@inproceedings{Tcl-2010-Ellogon,
    title        = {{E}llogon and the challenge of threads},
    author       = {Petasis, Georgios},
    year         = 2010,
    month        = {October 11--15},
    booktitle    = {Proceedings of the 17th Annual Tcl/Tk Conference (Tcl 2010)},
    address      = {Hilton Suites Chicago/Oakbrook Terrace, 10 Drury Lane, Oakbrook Terrace, Illinois, United States 60181},
    url          = {http://www.ellogon.org/petasis/bibliography/Tcl2010/EllogonAndThreads.pdf},
    abstract     = {This paper is about the Ellogon language engineering platform, and the challenges faced in modernising it, in order to better exploit contemporary hardware. Ellogon is an open-source infrastructure, specialised in natural language processing. Following a data model that closely resembles TIPSTER, Ellogon can be used either as an autonomous application, offering a graphical user interface, or it can be embedded in a C/C++ application as a library. Ellogon has been implemented in C/C++ and Tcl/Tk: in fact Ellogon is a vanilla Tcl interpreter, with the Ellogon core loaded as a Tcl extension, and a set of Tcl/Tk scripts that implement the GUI. The core component of Ellogon, being a Tcl extension, heavily relies on Tcl objects to implement its data model, a decision made more than a decade ago, which poses difficulties into making Ellogon a multi-threaded application.}
}
@inproceedings{Tcl-2010-TileQtTileGTK,
    title        = {{T}ile{Q}t and {T}ile{G}tk: current status},
    author       = {Petasis, Georgios},
    year         = 2010,
    month        = {October 11--15},
    booktitle    = {Proceedings of the 17th Annual Tcl/Tk Conference (Tcl 2010)},
    address      = {Hilton Suites Chicago/Oakbrook Terrace, 10 Drury Lane, Oakbrook Terrace, Illinois, United States 60181},
    url          = {http://www.ellogon.org/petasis/bibliography/Tcl2010/TileQtAndTileGTK.pdf},
    abstract     = {This paper is about two Tile and Ttk themes, TileQt and TileGTK. Despite being two distinct and very different extensions, the motivation for their development was common: making Tk applications look as native as possible under the Linux operating system.}
}
@inproceedings{Tcl-2010-TkGecko,
    title        = {{T}k{G}ecko: {A}nother {A}ttempt for an {HTML} {R}enderer for {T}k},
    author       = {Petasis, Georgios},
    year         = 2010,
    month        = {October 11--15},
    booktitle    = {Proceedings of the 17th Annual Tcl/Tk Conference (Tcl 2010)},
    address      = {Hilton Suites Chicago/Oakbrook Terrace, 10 Drury Lane, Oakbrook Terrace, Illinois, United States 60181},
    url          = {http://www.ellogon.org/petasis/bibliography/Tcl2010/TkGecko.pdf},
    abstract     = {The support for displaying HTML and especially complex Web sites has always been problematic in Tk. Several efforts have been made in order to alleviate this problem, and this paper presents another (and still incomplete) one. This paper presents TkGecko, a Tcl/Tk extension written in C++, which allows Gecko (the HTML processing and rendering engine developed by the Mozilla Foundation) to be embedded as a widget in Tk. The current status of the TkGecko extension is alpha quality, while the code is publically available under the BSD license.}
}
@inproceedings{Tcl2010-TkRibbon,
    title        = {{T}k{R}ibbon: {W}indows {R}ibbons for {T}k},
    author       = {Petasis, Georgios},
    year         = 2010,
    month        = {October 11--15},
    booktitle    = {Proceedings of the 17th Annual Tcl/Tk Conference (Tcl 2010)},
    address      = {Hilton Suites Chicago/Oakbrook Terrace, 10 Drury Lane, Oakbrook Terrace, Illinois, United States 60181},
    url          = {http://www.ellogon.org/petasis/bibliography/Tcl2010/TkRibbon.pdf},
    abstract     = {This paper is about TkRibbon, a Tcl/Tk extension that aims to introduce support for the Windows Ribbon Framework in the Tk toolkit. The Windows Ribbon is a graphical interface where a set of toolbars are placed on tabs in a notebook widget, aiming to substitute traditional menus and toolbars. This paper briefly describes Windows Ribbon framework, the TkRibbon Tk extension and presents some examples on how TkRibbon can be used by Tk applications.}
}
@inproceedings{DBLP:conf/lrec/PetasisP10,
    title        = {{B}log{B}uster: {A} {T}ool for {E}xtracting {C}orpora from the {B}logosphere},
    author       = {Petasis, Georgios and Dimitrios Petasis},
    year         = 2010,
    month        = {May 17--23},
    booktitle    = {Proceedings of the 7th International Conference on Language Resources and Evaluation, LREC 2010},
    publisher    = {European Language Resources Association},
    address      = {Valletta, Malta},
    isbn         = {2-9517408-6-7},
    url          = {http://www.ellogon.org/petasis/bibliography/LREC2010/LREC2010-BlogBuster-CameraReady.pdf},
    abstract     = {This paper presents BlogBuster, a tool for extracting a corpus from the blogosphere. The topic of cleaning arbitrary web pages with the goal of extracting a corpus from web data, suitable for linguistic and language technology research and development, has attracted significant research interest recently. Several general purpose approaches for removing boilerplate have been presented in the literature; however the blogosphere poses additional requirements, such as a finer control over the extracted textual segments in order to accurately identify important elements, i.e. individual blog posts, titles, posting dates or comments. BlogBuster tries to provide such additional details along with boilerplate removal, following a rule-based approach. A small set of rules were manually constructed by observing a limited set of blogs from the Blogger and Wordpress hosting platforms. These rules operate on the DOM tree of an HTML page, as constructed by a popular browser, Mozilla Firefox. Evaluation results suggest that BlogBuster is very accurate when extracting corpora from blogs hosted in the Blogger and Wordpress, while exhibiting a reasonable precision when applied to blogs not hosted in these two popular blogging platforms.},
    editor       = {Nicoletta Calzolari and Khalid Choukri and Bente Maegaard and Joseph Mariani and Jan Odijk and Stelios Piperidis and Mike Rosner and Daniel Tapias}
}
@inproceedings{citeulike:9267249,
    title        = {{S}emi-automated ontology learning: the {BOEMIE} approach},
    author       = {Petasis, Georgios and Vangelis Karkaletsis and Anastasia Krithara and Georgios Paliouras and Constantine D. Spyropoulos},
    year         = 2009,
    month        = {June 1},
    day          = 1,
    booktitle    = {Proceedings of the First ESWC Workshop on Inductive Reasoning and Machine Learning on the Semantic Web (IRMLeS 2009), 6th European Semantic Web Conference (ESWC 2009)},
    address      = {Hersonissos, Crete, Greece},
    url          = {http://www.ellogon.org/petasis/bibliography/ESWC2009/IRMLeS2009-ESWC2009.pdf},
    abstract     = {In this paper we describe a semi-automated approach for ontology learning. Exploiting an ontology-based multimodal information extraction system, the ontology learning subsystem accumulates documents that are insufficiently analysed and through clustering proposes new concepts, relations and interpretation rules to be added to the ontology.},
    keywords     = {evolution, ontologies}
}
@article{Castano01102009,
    title        = {{M}ultimedia {I}nterpretation for {D}ynamic {O}ntology {E}volution},
    author       = {Silvana Castano and Irma Sofia Espinosa Peraldi and Alfio Ferrara and Karkaletsis, Vangelis and Atila Kaya and Ralf M{\"o}ller and Stefano Montanelli and Petasis, Georgios and Michael Wessel},
    year         = 2009,
    journal      = {Journal of Logic and Computation},
    volume       = 19,
    number       = 5,
    pages        = {859--897},
    doi          = {10.1093/logcom/exn049},
    url          = {http://logcom.oxfordjournals.org/content/19/5/859.abstract},
    abstract     = {The recent success of distributed and dynamic infrastructures for knowledge sharing has raised the need for semiautomatic/automatic ontology evolution strategies. Ontology evolution is generally defined as the timely adaptation of an ontology to changing requirements and the consistent propagation of changes to dependent artifacts. In this article, we present an ontology evolution approach in the context of multimedia interpretation. Ontology evolution in this context relies on the results obtained through reasoning for the interpretation of multimedia resources, through population of the ontology with new individuals or through enrichment of the ontology with new concepts and new semantic relations. The article analyses the results of interpretation, population and enrichment obtained in evaluation experiments in terms of measures such as precision and recall. The evaluation reveals encouraging results.},
    eprint       = {http://logcom.oxfordjournals.org/content/19/5/859.full.pdf+html}
}
@incollection{springerlink:10.1007/978-3-540-87391-4_66,
    title        = {{A} {F}ramework for {L}anguage-{I}ndependent {A}nalysis and {P}rosodic {F}eature {A}nnotation of {T}ext {C}orpora},
    author       = {Dimitris Spiliotopoulos and Petasis, Georgios and Georgios Kouroupetroglou},
    year         = 2008,
    month        = {September 8--12},
    booktitle    = {Text, Speech and Dialogue (Proceedings of the 11th International Conference on Text, Speech and Dialogue (TSD 2008))},
    publisher    = {Springer Berlin / Heidelberg},
    address      = {Brno, Czech Republic},
    series       = {Lecture Notes in Computer Science},
    volume       = 5246,
    pages        = {517--524},
    doi          = {10.1007/978-3-540-87391-4_66},
    isbn         = {978-3-540-87390-7},
    url          = {http://dx.doi.org/10.1007/978-3-540-87391-4_66},
    abstract     = {Concept-to-Speech systems include Natural Language Generators that produce linguistically enriched text descriptions which can lead to significantly improved quality of speech synthesis. There are cases, however, where either the generator modules produce pieces of non-analyzed, non-annotated plain text, or such modules are not available at all. Moreover, the language analysis is restricted by the usually limited domain coverage of the generator due to its embedded grammar. This work reports on a language-independent framework basis, linguistic resources and language analysis procedures (word/sentence identification, part-of-speech, prosodic feature annotation) for text annotation/processing for plain or enriched text corpora. It aims to produce an automated XML- annotated enriched prosodic markup for English and Greek texts, for improved synthetic speech. The markup includes information for both training the synthesizer and for actual input for synthesising. Depending on the domain and target, different methods may be used for automatic classification of entities (words, phrases, sentences) to one or more preset categories such as {"}emphatic event{"}, {"}new/old information{"}, {"}second argument to verb{"}, {"}proper noun phrase{"}, etc. The prosodic features are classified according to the analysis of the speech-specific characteristics for their role in prosody modelling and passed through to the synthesizer via an extended SOLE-ML description. Evaluation results show that using selectable hybrid methods for part-of-speech tagging high accuracy is achieved. Annotation of a large generated text corpus containing 50{\%} enriched text and 50{\%} canned plain text produces a fully annotated uniform SOLE-ML output containing all prosodic features found in the initial enriched source. Furthermore, additional automatically-derived prosodic feature annotation and speech synthesis related values are assigned, such as word-placement in sentences and phrases, previous and next word entity relations, emphatic phrases containing proper nouns, and more.},
    editor       = {Sojka, Petr and Hor\'{a}k, Ale\v{s} and Kope\v{c}ek, Ivan and Pala, Karel}
}
@inproceedings{citeulike:5663452,
    title        = {{S}egmenting {HTML} pages using visual and semantic information},
    author       = {Petasis, Georgios and Pavlina Fragkou and Aris Theodorakos and Vangelis Karkaletsis and Constantine D. Spyropoulos},
    year         = 2008,
    month        = {June 1},
    journal      = {4th Web as Corpus Workshop (WAC-4)},
    booktitle    = {Proceedings of the 4th Web as a Corpus Workshop (WAC-4), 6th Language Resources and Evaluation Conference (LREC 2008)},
    address      = {Marrakech, Morocco},
    pages        = {18--24},
    doi          = {10.1109/SPCA.2006.297506},
    url          = {http://www.ellogon.org/petasis/bibliography/LREC2008/LREC-2008-SemanticSegmentation-Submitted.pdf},
    note         = {Proceedings: The 4th Web as Corpus: Can we do better than Google? http://www.lrec-conf.org/proceedings/lrec2008/workshops/W19_Proceedings.pdf},
    abstract     = {The information explosion of the Web aggravates the problem of effective information retrieval. Even though linguistic approaches found in the literature perform linguistic annotation by creating metadata in the form of tokens, lemmas or part of speech tags, however,this process is insufficient. This is due to the fact that these linguistic metadata do not exploit the actual content of the page, leading to the need of performing semantic annotation based on a predefined semantic model. This paper proposes a new learning approach for performing automatic semantic annotation. This is the result of a two step procedure: the first step partitions a web page into blocks based on its visual layout, while the second, performs subsequent partitioning based on the examination of appearance of specific types of entities denoting the semantic category as well as the application of a number of simple heuristics. Preliminary experiments performed on a manually annotated corpus regarding athletics proved to be very promising.}
}
@inproceedings{DBLP:conf/lrec/FragkouPTKS08,
    title        = {{BOEMIE} {O}ntology-{B}ased {T}ext {A}nnotation {T}ool},
    author       = {Pavlina Fragkou and Petasis, Georgios and Aris Theodorakos and Karkaletsis, Vangelis and Constantine D. Spyropoulos},
    year         = 2008,
    month        = {May 26 -- June 1},
    booktitle    = {Proceedings of the 6th International Conference on Language Resources and Evaluation (LREC 2008)},
    publisher    = {European Language Resources Association},
    address      = {Marrakech, Morocco},
    url          = {http://www.ellogon.org/petasis/bibliography/LREC2008/LREC-2008-324_paper.pdf},
    abstract     = {The huge amount of the available information in the Web creates the need of effective information extraction systems that are able to produce metadata that satisfy users information needs. The development of such systems, in the majority of cases, depends on the availability of an appropriately annotated corpus in order to learn extraction models. The production of such corpora can be significantly facilitated by annotation tools that are able to annotate, according to a defined ontology, not only named entities but most importantly relations between them. This paper describes the BOEMIE ontology-based annotation tool which is able to locate blocks of text that correspond to specific types of named entities, fill tables corresponding to ontology concepts with those named entities and link the filled tables based on relations defined in the domain ontology. Additionally, it can perform annotation of blocks of text that refer to the same topic. The tool has a user-friendly interface, supports automatic pre-annotation, annotation comparison as well as customization to other annotation schemata. The annotation tool has been used in a large scale annotation task involving 3000 web pages regarding athletics. It has also been used in another annotation task involving 503 web pages with medical information, in different languages.}
}
@inproceedings{Petasis:2008:LCG:1567281.1567350,
    title        = {{L}earning context-free grammars to extract relations from text},
    author       = {Petasis, Georgios and Karkaletsis, Vangelis and Georgios Paliouras and Constantine D. Spyropoulos},
    year         = 2008,
    booktitle    = {Proceeding of the 2008 conference on ECAI 2008: 18th European Conference on Artificial Intelligence},
    publisher    = {IOS Press},
    address      = {Amsterdam, The Netherlands, The Netherlands},
    series       = {Frontiers in Artificial Intelligence and Applications},
    volume       = 178,
    pages        = {303--307},
    isbn         = {978-1-58603-891-5},
    url          = {http://www.ellogon.org/petasis/bibliography/ECAI2008/ECAI2008_0371.pdf},
    abstract     = {In this paper we propose a novel relation extraction method, based on grammatical inference. Following a semi-supervised learning approach, the text that connects named entities in an annotated corpus is used to infer a context free grammar. The grammar learning algorithm is able to infer grammars from positive examples only, controlling overgeneralisation through minimum description length. Evaluation results show that the proposed approach performs comparable to the state of the art, while exhibiting a bias towards precision, which is a sign of conservative generalisation.},
    editor       = {Malik Ghallab and Constantine D. Spyropoulos and Nikos Fakotakis and Nikolaos M. Avouris}
}
@inproceedings{IWOD07,
    title        = {{O}ntology {D}ynamics with {M}ultimedia {I}nformation: {T}he {BOEMIE} {E}volution {M}ethodology},
    author       = {Silvana Castano and Sofia Espinosa and Alfio Ferrara and Karkaletsis, Vangelis and Atila Kaya and Sylvia Melzer and Ralf Moller and Stefano Montanelli and Petasis, Georgios},
    year         = 2007,
    month        = {June 7},
    booktitle    = {Proceedings of the International ESWC Workshop on Ontology Dynamics (IWOD 2007)},
    address      = {Innsbruck, Austria},
    url          = {http://www.ellogon.org/petasis/bibliography/IWOD2007/IWOD2007-paper-07.pdf},
    note         = {http://kmi.open.ac.uk/events/iwod/},
    abstract     = {In this paper, we present the ontology evolution methodology developed in the context of the BOEMIE project. Ontology evolution in BOEMIE relies on the results obtained through reasoning for the interpretation of multimedia resources in order to evolve (enhance) the ontology, through population of the ontology with new instances, or through enrichment of the ontology with new concepts and new semantic relations.}
}
@inproceedings{SPECOM-2005-Spiliotopoulos,
    title        = {{P}rosodically {E}nriched {T}ext {A}nnotation for {H}igh {Q}uality {S}peech {S}ynthesis},
    author       = {Dimitris Spiliotopoulos and Petasis, Georgios and Georgios Kouroupetroglou},
    year         = 2005,
    month        = {October 17--19},
    booktitle    = {Proceedings of the 10th International Conference on Speech and Computer (SPECOM-2005)},
    address      = {Patras, Greece},
    pages        = {313--316},
    url          = {http://www.ellogon.org/petasis/bibliography/SPECOM2005/Spiliotopoulos-SPECOM-2005.pdf},
    abstract     = {Linguistically enriched text generated from natural language modules contributes significantly on the quality of speech synthesis. For all cases where such modules are not available, such enriched input needs to be produced from plain text in order to maintain quality. This work reports on a framework of several combined language resources and procedures (word/sentence identification, syntactic analysis, prosodic feature annotation) for text annotation/processing from plain text. Using that, the implementation of an automatic XML formatted output generation module produces the prosodically enriched markup.}
}
@inproceedings{DBLP:conf/icgi/PetasisPSH04,
    title        = {{E}g-{GRIDS}: {C}ontext-{F}ree {G}rammatical {I}nference from {P}ositive {E}xamples {U}sing {G}enetic {S}earch},
    author       = {Petasis, Georgios and Georgios Paliouras and Constantine D. Spyropoulos and Constantine Halatsis},
    year         = 2004,
    month        = {October 11--13},
    booktitle    = {Grammatical Inference: Algorithms and Applications, Proceedings of the 7th International Colloquium on Grammatical Inference (ICGI 2004)},
    publisher    = {Springer Berlin / Heidelberg},
    address      = {Athens, Greece},
    series       = {Lecture Notes in Computer Science},
    volume       = 3264,
    pages        = {223--234},
    isbn         = {3-540-23410-1},
    url          = {http://www.ellogon.org/petasis/bibliography/ICGI2004/e-GRIDS-ICGI-2004-Submission.pdf},
    abstract     = {In this paper we present eg-GRIDS, an algorithm for inducing context-free grammars that is able to learn from positive sample sentences. The presented algorithm, similar to its GRIDS predecessors, uses simplicity as a criterion for directing inference, and a set of operators for exploring the search space. In addition to the basic beam search strategy of GRIDS, eg-GRIDS incorporates an evolutionary grammar selection process, aiming to explore a larger part of the search space. Evaluation results are presented on artificially generated data, comparing the performance of beam search and genetic search. These results show that genetic search performs better than beam search while being significantly more efficient computationally.},
    editor       = {Georgios Paliouras and Yasubumi Sakakibara}
}
@inproceedings{DBLP:conf/ecai/PetasisKGHPVC04,
    title        = {{A}daptive, {M}ultilingual {N}amed {E}ntity {R}ecognition in {W}eb {P}ages},
    author       = {Petasis, Georgios and Karkaletsis, Vangelis and Claire Grover and Ben Hachey and Maria Teresa Pazienza and Michele Vindigni and Jos{\'e} Coch},
    year         = 2004,
    month        = {August 22--27},
    booktitle    = {Proceedings of the 16th Eureopean Conference on Artificial Intelligence (ECAI'2004), including Prestigious Applicants of Intelligent Systems (PAIS 2004)},
    publisher    = {IOS Press},
    address      = {Valencia, Spain},
    pages        = {1073--1074},
    isbn         = {1-58603-452-9},
    url          = {http://www.ellogon.org/petasis/bibliography/ECAI2004/Petasis-ECAI2004-Poster.pdf},
    note         = {Extended version: http://www.ellogon.org/petasis/bibliography/ECAI2004/ECAI2004_NERC.pdf},
    abstract     = {Most of the information on the Web today is in the form of HTML documents, which are designed for presentation purposes and not for machine understanding and reasoning. Existing web extraction systems require a lot of human involvement for maintenance due to changes to targeted web sites and for adaptation to new web sites or even to new domains. This paper presents the adaptive, multilingual named entity recognition and classification (NERC) technologies developed for processing web pages in the context of the R{\&}D project CROSSMARC. The evaluation results demonstrate the viability of our approach.},
    crossref     = {DBLP:conf/ecai/2004},
    editor       = {Ramon L{\'o}pez de M{\'a}ntaras and Lorenza Saitta}
}
@inproceedings{DBLP:conf/setn/PerantonisGMKP04,
    title        = {{T}ext {A}rea {I}dentification in {W}eb {I}mages},
    author       = {Stavros J. Perantonis and Basilios Gatos and Vassilios Maragos and Karkaletsis, Vangelis and Petasis, Georgios},
    year         = 2004,
    month        = {May},
    booktitle    = {Methods and Applications of Artificial Intelligence, Proceedings of the 3rd Hellenic Conference on Artificial Intelligence (SETN 2004)},
    publisher    = {Springer Berlin / Heidelberg},
    address      = {Samos, Greece},
    series       = {Lecture Notes in Computer Science},
    volume       = 3025,
    pages        = {82--92},
    isbn         = {3-540-21937-4},
    url          = {http://www.ellogon.org/petasis/bibliography/SETN2004/SETN2004.pdf},
    abstract     = {With the explosive growth of the World Wide Web, millions of documents are published and accessed on-line. Statistics show that a significant part of Web text information is encoded in Web images. Since Web images have special characteristics that sometimes distinguish them from other types of images, commercial OCR products often fail to recognize Web images due to their special characteristics. This paper proposes a novel Web image processing algorithm that aims to locate text areas and prepare them for OCR procedure with better results. Our methodology for text area identification has been fully integrated with an OCR engine and with an Information Extraction system. We present quantitative results for the performance of the OCR engine as well as qualitative results concerning its effects to the Information Extraction system. Experimental results obtained from a large corpus of Web images, demonstrate the efficiency of our methodology.},
    editor       = {George A. Vouros and Themis Panayiotopoulos}
}
@article{GRAMMARS-vol.7-Petasis,
    title        = {{E}-{GRIDS}: {C}omputationally {E}fficient {G}rammatical {I}nference from {P}ositive {E}xamples},
    author       = {Petasis, Georgios and Georgios Paliouras and Vangelis Karkaletsis and Constantine Halatsis and Constantine D. Spyropoulos},
    year         = 2004,
    journal      = {GRAMMARS},
    volume       = 7,
    pages        = {69--110},
    url          = {http://www.ellogon.org/petasis/bibliography/GRAMMARS/GRAMMARS2004.pdf},
    note         = {Technical Report referenced in the paper: http://www.ellogon.org/petasis/bibliography/GRAMMARS/GRAMMARS2004-SpecialIssue-Petasis-TechnicalReport.pdf},
    abstract     = {In this paper we present a new computationally efficient algorithm for inducing context-free grammars that is able to learn from positive sample sentences. This new algorithm uses simplicity as a criterion for directing inference, and the search process of the new algorithm has been optimised by utilising the results of a theoretical analysis regarding the behaviour and complexity of the search operators. Evaluation results are presented on artificially generated data, while the scalability of the algorithm is tested on a large textual corpus. These results show that the new algorithm performs well and can infer grammars from large data sets in a reasonable amount of time.},
    keywords     = {grammatical inference, context-free grammars, minimum description length, positive examples}
}
@inproceedings{BCI2003-Petasis,
    title        = {{U}sing the {E}llogon {N}atural {L}anguage {E}ngineering {I}nfrastructure},
    author       = {Petasis, Georgios and Vangelis Karkaletsis and Georgios Paliouras and Constantine D. Spyropoulos},
    year         = 2003,
    month        = {November 21},
    booktitle    = {Proceedings of the Workshop on Balkan Language Resources and Tools, 1st Balkan Conference in Informatics (BCI 2003)},
    address      = {Thessaloniki, Greece},
    url          = {http://www.ellogon.org/petasis/bibliography/BCI2003/BCI2003-Petasis.pdf},
    note         = {http://labs-repos.iit.demokritos.gr/skel/bci03_workshop/},
    abstract     = {Ellogon is a multi-lingual, cross-operating system, general-purpose natural language engineering infrastructure. Ellogon has been used extensively in various NLP applications. It is currently provided for free for research use to research and academic organisations. In this paper, we outline its architecture and data model, present Ellogon features as used by different types of users and discuss its functionalities against other infrastructures for language engineering.}
}
@inproceedings{RANLP2003-Petasis,
    title        = {{C}ross-lingual {I}nformation {E}xtraction from {W}eb pages: the use of a general-purpose {T}ext {E}ngineering {P}latform},
    author       = {Petasis, Georgios and Vangelis Karkaletsis and Constantine D. Spyropoulos},
    year         = 2003,
    month        = {September 10--12},
    booktitle    = {Proceedings of the 4th International Conference on Recent Advances in Natural Language Processing (RANLP 2003)},
    address      = {Borovets, Bulgaria},
    pages        = {381--388},
    url          = {http://www.ellogon.org/petasis/bibliography/RANLP2003/RANLP-CameraReady.pdf},
    note         = {http://lml.bas.bg/ranlp2003/},
    abstract     = {In this paper we present how the use of a general-purpose text engineering platform has facilitated the development of a cross-lingual information extraction system and its adaptation to new domains and languages. Our approach for crosslingual information extraction from the Web covers all the way from the identification of Web sites of interest, to the location of the domain specific Web pages, to the extraction of specific information from the Web pages and its presentation to the end-user. This approach has been implemented in the context of the IST project CROSSMARC. The text engineering platform {"}Ellogon{"} offers functionalities that facilitated the development of core CROSSMARC components as well as their porting into new domains and languages.}
}
@incollection{springerlink:10.1007/3-540-38076-0_26,
    title        = {{A} {G}reek {M}orphological {L}exicon and {I}ts {E}xploitation by {N}atural {L}anguage {P}rocessing {A}pplications},
    author       = {Petasis, Georgios and Karkaletsis, Vangelis and Dimitra Farmakiotou and Ion Androutsopoulos and Constantine D. Spyropoulos},
    year         = 2003,
    booktitle    = {Advances in Informatics - Post-proceedings of the 8th Panhellenic Conference in Informatics},
    publisher    = {Springer Berlin / Heidelberg},
    series       = {Lecture Notes in Computer Science},
    volume       = 2563,
    pages        = {401--419},
    doi          = {10.1007/3-540-38076-0_26},
    isbn         = {3-540-07544-5},
    url          = {http://www.ellogon.org/petasis/bibliography/PCI2003/25630398.pdf},
    note         = {http://www.springerlink.com/content/hcdjrlvj5nlybf5c/},
    abstract     = {This paper presents a large-scale Greek morphological lexicon, developed at the Software {\&} Knowledge Engineering Laboratory (SKEL) of NCSR {"}Demokritos{"}. The paper describes the lexicon architecture and the procedure to develop and update it. The morphological lexicon was used to develop a lemmatiser and a morphological analyser that were exploited in various natural language processing applications for Greek. The paper presents these applications (controlled language checker, information extraction, information filtering) and discusses further research issues and how we plan to address them.},
    editor       = {Yannis Manolopoulos and Skevos Evripidou and Antonis Kakas}
}
@inproceedings{FarmakiotouEtAl02,
    title        = {{P}at{E}dit: {A}n {I}nformation {E}xtraction {P}attern {E}ditor for {F}ast {S}ystem {C}ustomization},
    author       = {Dimitra Farmakiotou and Karkaletsis, Vangelis and Ioannis Koutsias and Petasis, Georgios and Constantine D. Spyropoulos},
    year         = 2002,
    month        = {May 29--31},
    booktitle    = {Proceedings of the 3rd International Conference on Language Resources and Evaluation (LREC 2002)},
    publisher    = {European Language Resources Association},
    address      = {Las Palmas, Canary Islands, Spain},
    pages        = {1097--1102},
    url          = {http://www.ellogon.org/petasis/bibliography/LREC2002/LREC2002_Farmakiotou.pdf},
    abstract     = {This paper addresses the problem of Information Extraction (IE) system customization to new domains and extraction needs with the use of PatEdit, an IE Pattern Editor. PatEdit is a human-assisted knowledge engineering tool, that facilitates the production of IE patterns. First, we present the problem of IE system customisation and the use of human assisted knowledge engineering tools. Then, we describe PatEdit with respect to the IE pattern language used and discuss its characteristics that facilitate rapid pattern writing. Finally, the exploitation of PatEdit in two information extraction projects is presented along with our plans for future work.}
}
@inproceedings{LREC2002-Grover,
    title        = {{M}ultilingual {XML}-{B}ased {N}amed {E}ntity {R}ecognition for {E}-{R}etail {D}omains},
    author       = {Claire Grover and Scott Mcdonald and Donnla Nic Gearailt and Vangelis Karkaletsis and Dimitra Farmakiotou and Georgios Samaritakis and Petasis, Georgios and Maria Teresa Pazienza and Michele Vindigni and Frantz Vichot},
    year         = 2002,
    month        = {May 29--31},
    booktitle    = {Proceedings of the 3rd International Conference on Language Resources and Evaluation (LREC 2002)},
    publisher    = {European Language Resources Association},
    address      = {Las Palmas, Canary Islands, Spain},
    url          = {http://www.ellogon.org/petasis/bibliography/LREC2002/LREC2002_Grover.pdf},
    abstract     = {We describe the multilingual Named Entity Recognition and Classification (NERC) subpart of an e-retail product comparison system which is currently under development as part of the EU-funded project CROSSMARC. The system must be rapidly extensible, both to new languages and new domains. To achieve this aim we use XML as our common exchange format and the monolingual NERC components use a combination of rule-based and machine-learning techniques. It has been challenging to process web pages which contain heavily structured data where text is intermingled with HTML and other code. Our preliminary evaluation results demonstrate the viability of our approach.}
}
@inproceedings{Petasis02ellogon:a,
    title        = {{E}llogon: {A} {N}ew {T}ext {E}ngineering {P}latform},
    author       = {Petasis, Georgios and Karkaletsis, Vangelis and Georgios Paliouras and Ion Androutsopoulos and Constantine D. Spyropoulos},
    year         = 2002,
    month        = {May 29--31},
    booktitle    = {Proceedings of the 3rd International Conference on Language Resources and Evaluation (LREC 2002)},
    publisher    = {European Language Resources Association},
    address      = {Las Palmas, Canary Islands, Spain},
    pages        = {72--78},
    url          = {http://www.ellogon.org/petasis/bibliography/LREC2002/LREC2002_Petasis.pdf},
    abstract     = {This paper presents Ellogon, a multi-lingual, cross-platform, general-purpose text engineering environment. Ellogon was designed in order to aid both researchers in natural language processing, as well as companies that produce language engineering systems for the end-user. Ellogon provides a powerful TIPSTER-based infrastructure for managing, storing and exchanging textual data, embedding and managing text processing components as well as visualising textual data and their associated linguistic information. Among its key features are full Unicode support, an extensive multi-lingual graphical user interface, its modular architecture and the reduced hardware requirements.}
}
@inproceedings{DBLP:conf/setn/AndroutsopoulosSSDKS02,
    title        = {{N}amed {E}ntity {R}ecognition from {G}reek {W}eb {P}ages},
    author       = {Dimitra Farmakiotou and Karkaletsis, Vangelis and Georgios Samaritakis and Petasis, Georgios and Constantine D. Spyropoulos},
    year         = 2002,
    month        = {April 11--12},
    booktitle    = {Proceedings of the 2nd Hellenic Conference on Artificial Intelligence (SETN-02), Companion Volume},
    address      = {Thessaloniki, Greece},
    pages        = {91--102},
    url          = {http://www.ellogon.org/petasis/bibliography/SETN2002/091.pdf},
    note         = {http://lpis.csd.auth.gr/setn02/},
    abstract     = {We describe the functionalities of the Hellenic Named Entity Recognition and Classification (HNERC) system developed in the context of the CROSSMARC project. CROSSMARC is developing technology for e-retail product comparison. The CROSSMARC system locates relevant retailers web pages and processes them in order to extract information about their products (e.g. technical features, prices). CROSSMARCs technology is demonstrated and evaluated for two different product types and four languages (English, Greek, Italian, French). This paper presents the HNERC system that is responsible for the identification and classification of specific types of proper names (e.g. laptop manufacturers, models), numerical expressions (e.g. length, weight), and temporal expressions (e.g. time, date) in Hellenic vendor sites. The paper presents the HNERC processing stages using examples from the laptops domain.},
    editor       = {Ioannis P. Vlahavas and Constantine D. Spyropoulos}
}
@incollection{Petasis:2002:SNL:647292.722672,
    title        = {{S}ymbolic and {N}eural {L}earning of {N}amed-{E}ntity {R}ecognition and {C}lassification {S}ystems in {T}wo {L}anguages},
    author       = {Petasis, Georgios and Sergios Petridis and Georgios Paliouras and Karkaletsis, Vangelis and Stavros J. Perantonis and Constantine D. Spyropoulos},
    year         = 2002,
    month        = {January},
    booktitle    = {Advances in Computational Intelligence and Learning: Methods and Applications},
    publisher    = {Springer Berlin / Heidelberg},
    series       = {International Series in Intelligent Technologies},
    volume       = 18,
    pages        = {193--210},
    isbn         = {978-0-7923-7645-3},
    url          = {http://www.ellogon.org/petasis/bibliography/COIL2000/COILBook2001.pdf},
    note         = {http://www.springer.com/mathematics/book/978-0-7923-7645-3},
    abstract     = {This paper compares two alternative approaches to the problem of acquiring named-entity recognition and classification systems from training corpora, in two different languages. The process of named-entity recognition and classification is an important subtask in most language engineering applications, in particular information extraction, where different types of named entity are associated with specific roles in events. The manual construction of rules for the recognition of named entities is a tedious and time-consuming task. For this reason, effective methods to acquire such systems automatically from data are very desirable. In this paper we compare two popular learning methods on this task: a decision-tree induction method and a multi-layered feed-forward neural network. Particular emphasis is paid on the selection of the appropriate data representation for each method and the extraction of training examples from unstructured textual data. We compare the performance of the two methods on large corpora of English and Greek texts and present the results. In addition to the good performance of both methods, one very interesting result is the fact that a simple representation of the data, which ignores the order of the words within a named entity, leads to improved results over a more complex approach that preserves word order.},
    editor       = {Hans-Jurgen Zimmermann and Georgios Tselentis and Maarsten van Someren and Georgios Dounias},
    keywords     = {named entity recognition, tree induction, neural networks}
}
@inproceedings{Petasis:2001:GML:1756269.1756295,
    title        = {{A} {G}reek {M}orphological {L}exicon and its {E}xploitation by a {G}reek {C}ontrolled {L}anguage {C}hecker},
    author       = {Petasis, Georgios and Karkaletsis, Vangelis and Dimitra Farmakiotou and Ion Androutsopoulos and Constantine D. Spyropoulos},
    year         = 2001,
    month        = {November 8--10},
    booktitle    = {Proceedings of the 8th Panhellenic Conference on Informatics (PCI'01)},
    series       = {PCI'01},
    pages        = {80--89},
    url          = {http://www.ellogon.org/petasis/bibliography/PCI2001/EPY-Morph-CameraReady.pdf},
    abstract     = {This paper presents a large-scale Greek morphological lexicon, developed by the Software {\&} Knowledge Engineering Laboratory (SKEL) of NCSR {"}Demokritos{"}. The paper describes the lexicon architecture and the procedure to develop and update it. The morphological lexicon was used to develop a lemmatiser and a morphological analyser that were included in a controlled language checker for Greek. The paper discusses the current coverage of the lexicon, as well as remaining issues and how we plan to address them. Our goal is to produce a wide-coverage morphological lexicon of Greek that can be easily exploited in several natural language processing applications.}
}
@inproceedings{Petasis:2001:UML:1073012.1073067,
    title        = {{U}sing {M}achine {L}earning to {M}aintain {R}ule-based {N}amed - {E}ntity {R}ecognition and {C}lassification {S}ystems},
    author       = {Petasis, Georgios and Frantz Vichot and Francis Wolinski and Georgios Paliouras and Karkaletsis, Vangelis and Constantine D. Spyropoulos},
    year         = 2001,
    month        = {July 9--11},
    booktitle    = {Proceedings of the 39th Annual Meeting on Association for Computational Linguistics},
    publisher    = {Association for Computational Linguistics},
    address      = {Toulouse, France},
    series       = {ACL '01},
    pages        = {426--433},
    doi          = {http://dx.doi.org/10.3115/1073012.1073067},
    url          = {http://www.ellogon.org/petasis/bibliography/ACL2001/ACL-2001-CameraReady.pdf},
    abstract     = {This paper presents a method that assists in maintaining a rule-based named-entity recognition and classification system. The underlying idea is to use a separate system, constructed with the use of machine learning, to monitor the performance of the rule-based system. The training data for the second system is generated with the use of the rule-based system, thus avoiding the need for manual tagging. The disagreement of the two systems acts as a signal for updating the rule-based system. The generality of the approach is illustrated by applying it to large corpora in two different languages: Greek and French. The results are very encouraging, showing that this alternative use of machine learning can assist significantly in the maintenance of rule-based systems.}
}
@inproceedings{NAACL2001-Karkaletsis,
    title        = {{A} {C}ontrolled {L}anguage {C}hecker {B}ased on the {E}llogon {T}ext {E}ngineering {P}latform},
    author       = {Vangelis Karkaletsis and Georgios Samaritakis and Petasis, Georgios and Dimitra Farmakiotou and Ion Androutsopoulos and Constantine D. Spyropoulos},
    year         = 2001,
    month        = {June 2--7},
    booktitle    = {Proceedings from Language Technologies 2001: The Second Meeting of the North American Chapter of the Association for Computational Linguistics (NAACL 2001)},
    address      = {Pittsburgh, PA, USA},
    pages        = {74--75},
    url          = {http://www.ellogon.org/petasis/bibliography/NAACL2001/NAACL01-demo-ABSTRACT.pdf},
    organization = {Carnegie Mellon University},
    keywords     = {controlled languages, Modern Greek}
}
@inproceedings{ECAI2000-Petasis,
    title        = {{L}earning {D}ecision {T}rees for {N}amed-{E}ntity {R}ecognition and {C}lassification},
    author       = {Georgios Paliouras and Karkaletsis, Vangelis and Petasis, Georgios and Constantine D. Spyropoulos},
    year         = 2000,
    month        = {August 20--25},
    booktitle    = {Proceedings of the 14th European Conference on Artificial Intelligence (ECAI 2000)},
    series       = {ECAI 2000},
    url          = {http://www.ellogon.org/petasis/bibliography/ECAI2000/ECAI-2000.pdf},
    abstract     = {We propose the use of decision tree induction as a solution to the problem of customising a named-entity recognition and classification (NERC) system to a specific domain. A NERC system assigns semantic tags to phrases that correspond to named entities, e.g. persons, locations and organisations. Typically, such a system makes use of two language resources: a recognition grammar and a lexicon of known names, classified by the corresponding named-entity types. NERC systems have been shown to achieve good results when the domain of application is very specific. However, the construction of the grammar and the lexicon for a new domain is a hard and time-consuming process. We propose the use of decision trees as NERC {"}grammars{"} and the construction of these trees using machine learning. In order to validate our approach, we tested C4.5 on the identification of person and organisation names involved in management succession events, using data from the sixth Message Understanding Conference. The results of the evaluation are very encouraging showing that the induced tree can outperform a grammar that was constructed manually.}
}
@inproceedings{TESTIA2000-Petasis,
    title        = {{M}achine {L}earning and {N}amed-{E}ntity {R}ecognition},
    author       = {Petasis, Georgios},
    year         = 2000,
    month        = {July 15--30},
    booktitle    = {Proceedings of the 8th ELSNET European Summer School on Language and Speech Communication on the subject of Text and Speech Triggered Information Access (TeSTIA 2000)},
    address      = {Chios, Greece}
}
@inproceedings{Petasis:2000:AAP:345508.345563,
    title        = {{A}utomatic adaptation of proper noun dictionaries through cooperation of machine learning and probabilistic methods},
    author       = {Petasis, Georgios and Alessandro Cucchiarelli and Paola Velardi and Georgios Paliouras and Karkaletsis, Vangelis and Constantine D. Spyropoulos},
    year         = 2000,
    month        = {July 24--28},
    booktitle    = {Proceedings of the 23rd Annual International ACM SIGIR Conference on Research and Development in Information Retrieval (SIGIR)},
    publisher    = {ACM},
    address      = {New York, NY, USA},
    series       = {SIGIR '00},
    pages        = {128--135},
    doi          = {http://doi.acm.org/10.1145/345508.345563},
    isbn         = {1-58113-226-3},
    url          = {http://www.ellogon.org/petasis/bibliography/SIGIR2000/SIGIR-CameraReady.pdf},
    abstract     = {The recognition of Proper Nouns (PNs) is considered an important task in the area of Information Retrieval and Extraction. However the high performance of most existing PN classifiers heavily depends upon the avail-ability of large dictionaries of domain-specific Proper Nouns, and a certain amount of manual work for rule writing or manual tagging. Though it is not a heavy requirement to rely on some existing PN dictionary (of-ten these resources are available on the web), its coverage of a domain corpus may be rather low, in absence of manual updating. In this paper we propose a technique for the automatic updating of a PN Dictionary through the cooperation of an inductive and a probabilistic classifier. In our experiments we show that, whenever an existing PN Dictionary allows the identification of 50{\%} of the proper nouns within a corpus, our technique allows, without additional manual effort, the successful recognition of about 90{\%} of the remaining 50{\%}.},
    keywords     = {information extraction, machine learning and IR, natural language processing for IR, text data mining}
}
@inproceedings{Petasis00c.:symbolic,
    title        = {{S}ymbolic and {N}eural {L}earning for {N}amed-{E}ntity {R}ecognition},
    author       = {Petasis, Georgios and Sergios Petridis and Georgios Paliouras and Karkaletsis, Vangelis and Stavros J. Perantonis and Constantine D. Spyropoulos},
    year         = 2000,
    month        = {June 19--23},
    booktitle    = {Proceedings of European Best Practice Workshops and Symposium on Computational Intelligence and Learning (COIL 2000)},
    address      = {Chios, Greece},
    pages        = {58--66},
    url          = {http://www.ellogon.org/petasis/bibliography/COIL2000/COIL-2000.pdf},
    abstract     = {Named-entity recognition involves the identification and classification of named entities in text. This is an important subtask in most language engineering applications, in particular information extraction, where different types of named entity are associated with specific roles in events. The manual construction of rules for the recognition of named entities is a tedious and time-consuming task. For this reason, we present in this paper two approaches to learning named-entity recognition rules from text. The first approach is a decision-tree induction method and the second a multi-layered feed-forward neural network. Particular emphasis is paid on the selection of the appropriate feature set for each method and the extraction of training examples from unstructured textual data. We compare the performance of the two methods on a large corpus of English text and present the results.},
    keywords     = {name entity recognition, tree induction, neural networks}
}
@incollection{HCI1999-Petasis,
    title        = {{U}sing {M}achine {L}earning {T}echniques for {P}art-{O}f-{S}peech {T}agging in the {G}reek {L}anguage},
    author       = {Petasis, Georgios and Georgios Paliouras and Karkaletsis, Vangelis and Constantine D. Spyropoulos and Ion Androutsopoulos},
    year         = 2000,
    month        = {May},
    booktitle    = {ADVANCES IN INFORMATICS: Proceedings of the 7th Hellenic Conference on Informatics (HCI '99)},
    publisher    = {World Scientific},
    pages        = {273--281},
    isbn         = {978-981-02-4192-6},
    url          = {http://www.ellogon.org/petasis/bibliography/HCI1999/EPY99.pdf},
    note         = {http://www.worldscibooks.com/compsci/4320.html},
    abstract     = {This article investigates the use of Transformation-Based Error-Driven learning for resolving part-of-speech ambiguity in the Greek language. The aim is not only to study the performance, but also to examine its dependence on different thematic domains. Results are presented here for two different test cases: a corpus on {"}management succession events{"} and a general-theme corpus. The two experiments show that the performance of this method does not depend on the thematic domain of the corpus, and its accuracy for the Greek language is around 95{\%}.},
    editor       = {Dimitrios I. Fotiadis and Stavros D. Nikolopoulos}
}
@article{Karkaletsis:1999:NRG:595358.595565,
    title        = {{N}amed-{E}ntity {R}ecognition from {G}reek and {E}nglish {T}exts},
    author       = {Karkaletsis, Vangelis and Georgios Paliouras and Petasis, Georgios and Natasa Manousopoulou and Constantine D. Spyropoulos},
    year         = 1999,
    month        = {October},
    journal      = {Journal of Intelligent and Robotic Systems},
    publisher    = {Kluwer Academic Publishers},
    address      = {Hingham, MA, USA},
    volume       = 26,
    number       = 2,
    pages        = {123--135},
    doi          = {10.1023/A:1008124406923},
    issn         = {0921-0296},
    url          = {http://www.ellogon.org/petasis/bibliography/JIRS1999/JIRS-1999.pdf},
    abstract     = {Named-entity recognition (NER) involves the identification and classification of named entities in text. This is an important subtask in most language engineering applications, in particular information extraction, where different types of named entity are associated with specific roles in events. In this paper, we present a prototype NER system for Greek texts that we developed based on a NER system for English. Both systems are evaluated on corpora of the same domain and of similar size. The time-consuming process for the construction and update of domain-specific resources in both systems led us to examine a machine learning method for the automatic construction of such resources for a particular application in a specific language.},
    keywords     = {information extraction, machine learning, named-entity recognition}
}
@inproceedings{ACAI1999-Petasis2,
    title        = {{E}xploiting {L}earning in {B}ilingual {N}amed {E}ntity {R}ecognition},
    author       = {Petasis, Georgios},
    year         = 1999,
    month        = {July 5--16},
    booktitle    = {Proceedings of the ECCAI Advanced Course on Artificial Intelligence (ACAI '99)},
    address      = {Chania, Greece},
    url          = {http://www.ellogon.org/petasis/bibliography/ACAI1999/ss1_07.pdf}
}
@inproceedings{Petasis99resolvingpart-of-speech,
    title        = {{R}esolving {P}art-of-{S}peech {A}mbiguity in the {G}reek {L}anguage {U}sing {L}earning {T}echniques},
    author       = {Petasis, Georgios and Georgios Paliouras and Karkaletsis, Vangelis and Constantine D. Spyropoulos},
    year         = 1999,
    month        = {July 5--16},
    booktitle    = {Proceedings of the ECCAI Advanced Course on Artificial Intelligence (ACAI '99)},
    address      = {Chania, Greece},
    url          = {http://www.ellogon.org/petasis/bibliography/ACAI1999/9906019.pdf},
    abstract     = {This article investigates the use of Transformation-Based Error-Driven learning for resolving part-of-speech ambiguity in the Greek language. The aim is not only to study the performance, but also to examine its dependence on different thematic domains. Results are presented here for two different test cases: a corpus on {"}management succession events{"} and a general-theme corpus. The two experiments show that the performance of this method does not depend on the thematic domain of the corpus, and its accuracy for the Greek language is around 95{\%}.}
}
@incollection{EURISCON1998-Karkaletsis,
    title        = {{N}amed {E}ntity {R}ecognition from {G}reek texts: the {GIE} {P}roject},
    author       = {Vangelis Karkaletsis and Constantine D. Spyropoulos and Petasis, Georgios},
    year         = 1999,
    booktitle    = {Advances in Intelligent Systems: Concepts, Tools and Applications},
    publisher    = {Springer Berlin / Heidelberg},
    series       = {Intelligent Systems, Control and Automation: Science and Engineering},
    volume       = 21,
    pages        = {131--142},
    isbn         = {978-1-4020-0393-6},
    url          = {http://www.springer.com/computer/image+processing/book/978-1-4020-0393-6},
    note         = {Presented at the 3rd European Robotics Intelligent Systems {\&} Control Conference (EURISCON '98), June 22--25 1998, Athens, Greece.},
    chapter      = 12,
    editor       = {Spyros G. Tzafestas},
    keywords     = {named-entity recognition, information extraction, machine learning}
}
