@inproceedings{konrad:24050:sign-lang:lrec,
  author    = {Konrad, Reiner and Hanke, Thomas and Isard, Amy and Schulder, Marc and K{\"o}nig, Lutz and Bleicken, Julian and B{\"o}se, Oliver},
  title     = {Corpus {\`a} la carte -- Improving Access to the {Public} {DGS} {Corpus}},
  pages     = {184--193},
  editor    = {Efthimiou, Eleni and Fotinea, Stavroula-Evita and Hanke, Thomas and Hochgesang, Julie A. and Mesch, Johanna and Schulder, Marc},
  booktitle = {Proceedings of the {LREC-COLING} 2024 11th Workshop on the Representation and Processing of Sign Languages: Evaluation of Sign Language Resources},
  maintitle = {2024 Joint International Conference on Computational Linguistics, Language Resources and Evaluation ({LREC-COLING} 2024)},
  publisher = {{ELRA Language Resources Association (ELRA) and the International Committee on Computational Linguistics (ICCL)}},
  address   = {Torino, Italy},
  day       = {25},
  month     = may,
  year      = {2024},
  isbn      = {978-2-493814-30-2},
  language  = {english},
  url       = {https://www.sign-lang.uni-hamburg.de/lrec/pub/24050.html},
  abstract  = {This article presents the fourth release of the Public DGS Corpus, a large corpus of German Sign Language (DGS). Since its first release in 2018, the Public DGS Corpus has provided its content through multiple portals to meet the needs of different user groups. Having started with a community portal and a research portal for general data access, the ANNIS portal for dynamic web-based exploration of the corpus was added in 2022. With this latest release, a fourth portal is added to allow sign language linguists to access the public corpus directly through the annotation software iLex. Furthermore, search capabilities and interconnectedness between the portals are strongly improved, allowing users to move between portals to combine their strengths. Additional improvements to the corpus include additional recordings, new pose information models, improved HamNoSys, enhanced type information and web interface revisions.}
}

@inproceedings{langer:18026:sign-lang:lrec,
  author    = {Langer, Gabriele and M{\"u}ller, Anke and W{\"a}hl, Sabrina},
  title     = {Queries and Views in {iLex} to Support Corpus-based Lexicographic Work on {German} {Sign} {Language} ({DGS})},
  pages     = {107--114},
  editor    = {Bono, Mayumi and Efthimiou, Eleni and Fotinea, Stavroula-Evita and Hanke, Thomas and Hochgesang, Julie A. and Kristoffersen, Jette and Mesch, Johanna and Osugi, Yutaka},
  booktitle = {Proceedings of the {LREC2018} 8th Workshop on the Representation and Processing of Sign Languages: Involving the Language Community},
  maintitle = {11th International Conference on Language Resources and Evaluation ({LREC} 2018)},
  publisher = {{European Language Resources Association (ELRA)}},
  address   = {Miyazaki, Japan},
  day       = {12},
  month     = may,
  year      = {2018},
  isbn      = {979-10-95546-01-6},
  language  = {english},
  url       = {https://www.sign-lang.uni-hamburg.de/lrec/pub/18026.html},
  abstract  = {In the DGS-Korpus project the corpus is being used as the basis for lexicographic descriptions of signs in dictionary entries. In this process the lexicographers start from the data and type entry structures as found in the annotation database. While preparing a dictionary entry much of the work consists of manually going through a number of single tokens viewing the original data and available annotations. Findings are then categorised and summarised. However, a number of decisions and descriptions are also supported by pre-defined searches and views on the data. Supported areas include lexicographic lemmatisation (lemma sign establishment), selection of citation forms and variants, grammatical behaviour of signs, collocational patterns of use, regional distribution patterns and distribution of lexical or formational variants over different age groups. While we are still in the process of exploring the possibilities of a sign language corpus for lexicography, searches and views that have proven useful for our work are exemplified in this paper with regard to dictionary entries.}
}

@inproceedings{boyesbraem:16008:sign-lang:lrec,
  author    = {Boyes Braem, Penny and Ebling, Sarah},
  title     = {Preventing Too Many Cooks from Spoiling the Broth: Some Questions and Suggestions for Collaboration between Projects in {iLex}},
  pages     = {25--28},
  editor    = {Efthimiou, Eleni and Fotinea, Stavroula-Evita and Hanke, Thomas and Hochgesang, Julie A. and Kristoffersen, Jette and Mesch, Johanna},
  booktitle = {Proceedings of the {LREC2016} 7th Workshop on the Representation and Processing of Sign Languages: Corpus Mining},
  maintitle = {10th International Conference on Language Resources and Evaluation ({LREC} 2016)},
  publisher = {{European Language Resources Association (ELRA)}},
  address   = {Portoro{\v z}, Slovenia},
  day       = {28},
  month     = may,
  year      = {2016},
  language  = {english},
  url       = {https://www.sign-lang.uni-hamburg.de/lrec/pub/16008.html},
  abstract  = {Collaborative development of sign language resources is fortunately becoming increasingly common. In the spirit of collaboration, having one shared lexicon for sign language projects is a big advantage. However, this poses challenges to aspects pertaining to consistency of data, privacy of informants, and intellectual property. This contribution points out some problems that arise, especially if the common data comes from projects of different institutions. We describe what we have found to be a sustainable legal framework for our collaborative iLex corpus lexicon, giving an overview of the different kinds of partners involved in the creation and exploitation of a shared iLex corpus lexicon and providing our answers to the questions we faced along with an outlook for the future.}
}

@inproceedings{ebling:16009:sign-lang:lrec,
  author    = {Ebling, Sarah and Boyes Braem, Penny},
  title     = {Linking a Web Lexicon of {DSGS} Technical Signs to {iLex}},
  pages     = {59--62},
  editor    = {Efthimiou, Eleni and Fotinea, Stavroula-Evita and Hanke, Thomas and Hochgesang, Julie A. and Kristoffersen, Jette and Mesch, Johanna},
  booktitle = {Proceedings of the {LREC2016} 7th Workshop on the Representation and Processing of Sign Languages: Corpus Mining},
  maintitle = {10th International Conference on Language Resources and Evaluation ({LREC} 2016)},
  publisher = {{European Language Resources Association (ELRA)}},
  address   = {Portoro{\v z}, Slovenia},
  day       = {28},
  month     = may,
  year      = {2016},
  language  = {english},
  url       = {https://www.sign-lang.uni-hamburg.de/lrec/pub/16009.html},
  abstract  = {A website for a lexicon of Swiss German Sign Language equivalents of technical terms was developed several years ago using Flash technology. In the intervening years, the backend research database was migrated from FileMaker to iLex. Here, we report on the development of a web platform that provides access to the same technical signs by extracting the relevant information directly from iLex. This new platform has many advantages: New sets of signs for technical terms can be added or existing ones modified in iLex at any time, and changes are reflected in the web platform upon refreshing the browser. Just as importantly, the new platform can now also be accessed through all major mobile operating systems, as it does not rely on Flash. We descri be how information on the glosses, keywords, videos of citation forms, status, and uses of the technical signs is represented in iLex and how the corresponding web platform was built.}
}

@inproceedings{hanke:16024:sign-lang:lrec,
  author    = {Hanke, Thomas},
  title     = {Towards a Visual Sign Language Corpus Linguistics},
  pages     = {89--92},
  editor    = {Efthimiou, Eleni and Fotinea, Stavroula-Evita and Hanke, Thomas and Hochgesang, Julie A. and Kristoffersen, Jette and Mesch, Johanna},
  booktitle = {Proceedings of the {LREC2016} 7th Workshop on the Representation and Processing of Sign Languages: Corpus Mining},
  maintitle = {10th International Conference on Language Resources and Evaluation ({LREC} 2016)},
  publisher = {{European Language Resources Association (ELRA)}},
  address   = {Portoro{\v z}, Slovenia},
  day       = {28},
  month     = may,
  year      = {2016},
  language  = {english},
  url       = {https://www.sign-lang.uni-hamburg.de/lrec/pub/16024.html},
  abstract  = {Visualisations have a long tradition in linguistics, as in many fields dealing with complex structure. New forms of representations have been introduced to Visual Linguistics in the recent past, e.g. to help the researcher find the needle in a haystack, i.e. corpus. Here we present visualisation services available in iLex making a combined corpus and lexical database visually accessible. While many approaches suggested for textual languages transfer to sign language data as well, others explore sign-specific structure, such as multi-dimensional concordances not being restricted to sequentiality. Experimental combinations of animated visualisation and image processing might support the researcher to compensate for incomplete high-quality (=manual) annotation. In the long run, we see the potential that visualisation and data manipulation go hand in hand, allowing future user interfaces that are less text-heavy than today's sign language annotation environments.}
}

@inproceedings{langer:16014:sign-lang:lrec,
  author    = {Langer, Gabriele and Troelsg{\aa}rd, Thomas and Kristoffersen, Jette and Konrad, Reiner and Hanke, Thomas and K{\"o}nig, Susanne},
  title     = {Designing a Lexical Database for a Combined Use of Corpus Annotation and Dictionary Editing},
  pages     = {143--152},
  editor    = {Efthimiou, Eleni and Fotinea, Stavroula-Evita and Hanke, Thomas and Hochgesang, Julie A. and Kristoffersen, Jette and Mesch, Johanna},
  booktitle = {Proceedings of the {LREC2016} 7th Workshop on the Representation and Processing of Sign Languages: Corpus Mining},
  maintitle = {10th International Conference on Language Resources and Evaluation ({LREC} 2016)},
  publisher = {{European Language Resources Association (ELRA)}},
  address   = {Portoro{\v z}, Slovenia},
  day       = {28},
  month     = may,
  year      = {2016},
  language  = {english},
  url       = {https://www.sign-lang.uni-hamburg.de/lrec/pub/16014.html},
  abstract  = {In a combined corpus-dictionary project, you would need one lexical database that could serve as a shared ``backbone'' for both corpus annotation and dictionary editing, but it is not that easy to define a database structure that applies satisfactorily to both these purposes. In this paper, we will exemplify the problem and present ideas on how to model structures in a lexical database that facilitate corpus annotation as well as dictionary editing. The paper is a joint work between the DGS Corpus Project and the DTS Dictionary Project. The two projects come from opposite sides of the spectrum (one adjusting a lexical database grown from dictionary making for corpus annotating, one building a lexical database in parallel with corpus annotation and editing a corpus-based dictionary), and we will consider requirements and feasible structures for a database that can serve both corpus and dictionary.}
}

@inproceedings{hanke:14029:sign-lang:lrec,
  author    = {Hanke, Thomas},
  title     = {Annotation of mouth activities with {iLex}},
  pages     = {67--70},
  editor    = {Crasborn, Onno and Efthimiou, Eleni and Fotinea, Stavroula-Evita and Hanke, Thomas and Hochgesang, Julie A. and Kristoffersen, Jette and Mesch, Johanna},
  booktitle = {Proceedings of the {LREC2014} 6th Workshop on the Representation and Processing of Sign Languages: Beyond the Manual Channel},
  maintitle = {9th International Conference on Language Resources and Evaluation ({LREC} 2014)},
  publisher = {{European Language Resources Association (ELRA)}},
  address   = {Reykjavik, Iceland},
  day       = {31},
  month     = may,
  year      = {2014},
  language  = {english},
  url       = {https://www.sign-lang.uni-hamburg.de/lrec/pub/14029.html},
  abstract  = {Recordings from the DGS-Korpus project with 330 informants confirm that at least for German Sign Language (DGS) you hardly find longer stretches of signing not accompanied by any mouth activity. Independent of whether you consider mouth activity while signing as part of the sign language proper or as a parallel system interacting with sign language to jointly transport meaning, mouth activity is part of the linguistic system used by signers and should be treated as such by any corpus approach. In a purely bottom-up approach an annotation practice used for mouth activities would try to describe the phenomena and leave it to a second step to classify (e.g. between mouthing and mouth gestures) and relate (e.g. to spoken language words). For practical reasons, however, the first step is often skipped, and separate coding systems are applied to what is categorised either as mouthing derived from spoken language or mouth gesture where there is no obvious connection between the meaning expressed and any spoken language words expressing that same meaning. This happens not only for time (=budget) reasons, but also because it is difficult for coders to describe mouth visemes precisely if the sign/mouth combo already suggests what is to be seen on the mouth. While there are established coding procedures to avoid influence as far as possible (like only showing the signer's face, provided video quality is good enough), they make the approach very time-consuming, even if not counting quality assurance measures like inter-transcriber agreement. Some projects undertaken at the IDGS in Hamburg therefore leave it with a spoken-language-driven approach: The mouth activity is classified as either mouth gesture or mouthing, and in the latter case the German word is noted down that a competent DGS signer ``reads'' from the lips, i.e. that word from the set of words to be expected with the co-temporal sign in its context that matches the observation. Standard orthography is used unless there is a substantial deviation. For mouth gestures, holistic labels are used. These two extremes span a whole spectrum of coding approaches that can be used for mouth activities. iLex, the Hamburg sign language annotation workbench, tries to support the whole range of solutions as good as possible. The poster w/ demo shows a variety of approaches actually in use or on the horizon and what iLex has to offer for each of those, from more time-series like systems to those evaluating co-occurrence and semantic relatedness, from novice-friendly decision trees to expert-only modes. Inter-transcriber agreement data on the examples given clearly show that a thorough analysis of data quality has to go beyond such measures.}
}

@inproceedings{hanke:12029:sign-lang:lrec,
  author    = {Hanke, Thomas and K{\"o}nig, Susanne and Konrad, Reiner and Langer, Gabriele},
  title     = {Towards tagging of multi-sign lexemes and other multi-unit structures},
  pages     = {67--68},
  editor    = {Crasborn, Onno and Efthimiou, Eleni and Fotinea, Stavroula-Evita and Hanke, Thomas and Kristoffersen, Jette and Mesch, Johanna},
  booktitle = {Proceedings of the {LREC2012} 5th Workshop on the Representation and Processing of Sign Languages: Interactions between Corpus and Lexicon},
  maintitle = {8th International Conference on Language Resources and Evaluation ({LREC} 2012)},
  publisher = {{European Language Resources Association (ELRA)}},
  address   = {Istanbul, Turkey},
  day       = {27},
  month     = may,
  year      = {2012},
  language  = {english},
  url       = {https://www.sign-lang.uni-hamburg.de/lrec/pub/12029.html},
  abstract  = {With the building of larger sign language corpora tagging, handling and analysing large amounts of data reach a new level of complexity. Efficiency and interpersonal consistency in tagging are relevant issues as well as procedures and structures to identify and tag relevant linguistic units and structures beyond and above the manual sign level. We present and discuss problems and possible solution approaches (focussing on the working environment of iLex) of how to deal with multi-unit structures and more specifically multi-sign lexemes in annotation and lexicon building.}
}

@inproceedings{hanke:10056:sign-lang:lrec,
  author    = {Hanke, Thomas and Storz, Jakob and Wagner, Sven},
  title     = {{iLex}: Handling Multi-Camera Recordings},
  pages     = {110--111},
  editor    = {Dreuw, Philippe and Efthimiou, Eleni and Hanke, Thomas and Johnston, Trevor and Mart{\'i}nez Ruiz, Gregorio and Schembri, Adam},
  booktitle = {Proceedings of the {LREC2010} 4th Workshop on the Representation and Processing of Sign Languages: Corpora and Sign Language Technologies},
  maintitle = {7th International Conference on Language Resources and Evaluation ({LREC} 2010)},
  publisher = {{European Language Resources Association (ELRA)}},
  address   = {Valletta, Malta},
  day       = {22--23},
  month     = may,
  year      = {2010},
  language  = {english},
  url       = {https://www.sign-lang.uni-hamburg.de/lrec/pub/10056.html},
  abstract  = {Until recently, sign language researchers were quite happy with just one or two views for each recording session. While ELAN allows the user to relate several media files to a transcript and to sync them, iLex just allows one single media container and relies on the container format, such as QuickTime, to group and sync several video streams into one container. In order to save screen real estate, iLex offers the user the possibility to switch on or off individual tracks within the media file. This works quite fine with two or three different views grouped, but fails to provide an adequate solution in multi-view projects such as Dicta-Sign or DGS Corpus with seven cameras altogether for a pair of informants. The advent of HD videos makes screen real estate really an issue: Even on very large screens, video competes with transcription space.
\par
Here we present a user interface study that allows flexible switching between video layouts whenever the transcription focus changes. Switching (including zooming and cropping) may be initiated at any point of time by the user, or can be automated to depend on tagging such as tasks or turns. This user interface is backed up by a server infrastructure providing videos in different spatial resolutions as needed for optimal display while saving transfer bandwidth and local processing power which even nowadays becomes an issue when dealing with several HD videos in parallel.}
}

@inproceedings{hanke:08011:sign-lang:lrec,
  author    = {Hanke, Thomas and Storz, Jakob},
  title     = {{iLex} -- A Database Tool for Integrating Sign Language Corpus Linguistics and Sign Language Lexicography},
  pages     = {64--67},
  editor    = {Crasborn, Onno and Efthimiou, Eleni and Hanke, Thomas and Thoutenhoofd, Ernst D. and Zwitserlood, Inge},
  booktitle = {Proceedings of the {LREC2008} 3rd Workshop on the Representation and Processing of Sign Languages: Construction and Exploitation of Sign Language Corpora},
  maintitle = {6th International Conference on Language Resources and Evaluation ({LREC} 2008)},
  publisher = {{European Language Resources Association (ELRA)}},
  address   = {Marrakech, Morocco},
  day       = {1},
  month     = jun,
  year      = {2008},
  language  = {english},
  url       = {https://www.sign-lang.uni-hamburg.de/lrec/pub/08011.html},
  abstract  = {This poster presents iLex, a software tool targeted at both corpus linguistics and lexicography. It is now a shared belief in the LR community that lexicographic work on any language should be based on a corpus. Conversely, lemmatisation of a sign language corpus requires a lexicon to be built up in parallel.
\par
For languages with a written form and orthography, lemmatisation is a more or less straight-forward process. For sign languages, however, type-token matching is a major task by itself. Glossing or form- based transcription, e.g. with HamNoSys, may be sufficient for small single-transcriber projects. Consistency, however, cannot be guaranteed over multiple transcribers, large quantities, or longer periods of time.
\par
iLex is therefore designed as a relational database linking tokens with their types. That means that the transcription process does not consist of assigning text tags to time intervals of the source video, but of tagging intervals with a reference to a type. The database then allows the user to review all tokens of a type at any point of time in order to verify that the intended type-token pair really fits with the type's definition and extension. Revisions of earlier decisions in the light of new data are as easy as dragging instances from one type to the other. Beyond the support in the initial type-token matching, iLex gives its users views onto the transcribed data orthogonal to the transcription itself, and thereby helps to improve transcription quality. With its ability to support users working on different projects in one database, iLex allows synergies between projects as each project immediately profits from data entered by others. The cost for these benefits is the necessity of a solid infrastructure: A database server needs to be installed, and ideally every user should have access to all videos, often requiring specialised video servers. For larger corpus projects, however, this should be taken for granted anyway. For data exchange with other research groups, iLex supports a number of file formats, such as ELAN, SignStream, and syncWRITER for transcription data and IMDI for metadata. While exporting data from iLex into these formats as well as a couple of presentation formats such as HTML with thumbnails is done with a simple menu command, importing data from other sources requires some additional steps to be done by the researcher. As other data formats consist of text tags only, some matching operations are necessary to convert from text to tokens. The newest release of iLex supports the user in this procedure: By learning a mapping from imported glosses to iLex types from user actions, it can partially automate future imports from the same source. In addition to data exchange with other transcription tools and export to presentation formats, iLex integrates with a number of tools for rapid production of sign language teaching materials and for virtual signing by means of avatars.
\par
On the lexicography side, iLex can host all the data necessary for the production of dictionaries. With its scripting language support, iLex is able to almost completely automate the production of a variety of formats including print, DVD, online websites for computers, and online websites for iPods/iPhones.}
}

@inproceedings{hanke:06014:sign-lang:lrec,
  author    = {Hanke, Thomas},
  title     = {Towards a Corpus-based Approach to Sign Language Dictionaries},
  pages     = {70--73},
  editor    = {Vettori, Chiara},
  booktitle = {Proceedings of the {LREC2006} 2nd Workshop on the Representation and Processing of Sign Languages: Lexicographic Matters and Didactic Scenarios},
  maintitle = {5th International Conference on Language Resources and Evaluation ({LREC} 2006)},
  publisher = {{European Language Resources Association (ELRA)}},
  address   = {Genoa, Italy},
  day       = {28},
  month     = may,
  year      = {2006},
  language  = {english},
  url       = {https://www.sign-lang.uni-hamburg.de/lrec/pub/06014.html},
  abstract  = {This paper discusses those aspects of iLex, a sign language transcription tool, that are relevant to lexical work and the production of e- learning materials. iLex is built upon a relational database, and uses this strength to support the user in type-token matching by giving immediate access to all other tokens already related to a certain type. iLex features a number of classification schemes, both built-in and data-driven, to allow for the incremental process of identifying and describing the lexicon of a sign language. Data cannot only be exported to other transcription tools, but also into authoring systems for teaching materials. Finally, we speculate about the applicability of Zipf's Law for sign language corpora extrapolating from the current contents of the iLex database.}
}

@inproceedings{hanke-2002-ilex:lrec,
  author    = {Hanke, Thomas},
  title     = {i{L}ex - A Tool for Sign Language Lexicography and Corpus Analysis},
  pages     = {923--926},
  editor    = {Rodr{\'i}guez, Manuel Gonz{\'a}lez and Araujo, Carmen Paz Suarez},
  booktitle = {3rd International Conference on Language Resources and Evaluation ({LREC} 2002)},
  publisher = {{European Language Resources Association (ELRA)}},
  address   = {Las Palmas, Canary Islands, Spain},
  day       = {27},
  month     = may,
  year      = {2002},
  language  = {english},
  url       = {https://aclanthology.org/L02-1330},
  abstract  = {This paper describes a tool that combines features found in empirical sign language lexicography and in sign language discourse transcription. It supports the user in lexicon building while working on the transcription of a corpus. While it tries to reach a certain level of compatibility with upcoming multimedia annotation tools, it offers a number of unique features considered essential due to the specific nature of sign languages.}
}

