@inproceedings{langer:16013:sign-lang:lrec,
  author    = {Langer, Gabriele and Hanke, Thomas and Konrad, Reiner and K{\"o}nig, Susanne},
  title     = {``Non-tokens'': When Tokens Should not Count as Evidence of Sign Use},
  pages     = {137--142},
  editor    = {Efthimiou, Eleni and Fotinea, Stavroula-Evita and Hanke, Thomas and Hochgesang, Julie A. and Kristoffersen, Jette and Mesch, Johanna},
  booktitle = {Proceedings of the {LREC2016} 7th Workshop on the Representation and Processing of Sign Languages: Corpus Mining},
  maintitle = {10th International Conference on Language Resources and Evaluation ({LREC} 2016)},
  publisher = {{European Language Resources Association (ELRA)}},
  address   = {Portoro{\v z}, Slovenia},
  day       = {28},
  month     = may,
  year      = {2016},
  language  = {english},
  url       = {https://www.sign-lang.uni-hamburg.de/lrec/pub/16013.html},
  abstract  = {Lemmatised corpora consist of tokens as instantiations of signs (types). Tokens usually count as evidences of the signs' use. Frequency of tokens is an important criterion for the lexical status of a sign. In combination with metadata on the signers' sociolinguistic backgrounds such as age, gender, and origin these tokens can also be analysed for regional and sociolinguistic variation. However, corpora may also contain instances of sign use that do not reflect the sign use of the person uttering them. This is particularly true for metalinguistic discussions of signs, malformed signing and slips of the hand as well as other phenomena such as copying/repeating signs of the interlocutors or from stimulus material. In our presentation we list and discuss different kinds of sign use (tokens) that should either not be counted as proof of a sign type at all or at least not as evidence of regular sign use by that particular person. Examples of these ``non-tokens'' are either taken from the DGS Corpus or from uploaded video answers of the DGS Feedback. We also discuss some implications on how to annotate these cases.}
}

@inproceedings{langer:16014:sign-lang:lrec,
  author    = {Langer, Gabriele and Troelsg{\aa}rd, Thomas and Kristoffersen, Jette and Konrad, Reiner and Hanke, Thomas and K{\"o}nig, Susanne},
  title     = {Designing a Lexical Database for a Combined Use of Corpus Annotation and Dictionary Editing},
  pages     = {143--152},
  editor    = {Efthimiou, Eleni and Fotinea, Stavroula-Evita and Hanke, Thomas and Hochgesang, Julie A. and Kristoffersen, Jette and Mesch, Johanna},
  booktitle = {Proceedings of the {LREC2016} 7th Workshop on the Representation and Processing of Sign Languages: Corpus Mining},
  maintitle = {10th International Conference on Language Resources and Evaluation ({LREC} 2016)},
  publisher = {{European Language Resources Association (ELRA)}},
  address   = {Portoro{\v z}, Slovenia},
  day       = {28},
  month     = may,
  year      = {2016},
  language  = {english},
  url       = {https://www.sign-lang.uni-hamburg.de/lrec/pub/16014.html},
  abstract  = {In a combined corpus-dictionary project, you would need one lexical database that could serve as a shared ``backbone'' for both corpus annotation and dictionary editing, but it is not that easy to define a database structure that applies satisfactorily to both these purposes. In this paper, we will exemplify the problem and present ideas on how to model structures in a lexical database that facilitate corpus annotation as well as dictionary editing. The paper is a joint work between the DGS Corpus Project and the DTS Dictionary Project. The two projects come from opposite sides of the spectrum (one adjusting a lexical database grown from dictionary making for corpus annotating, one building a lexical database in parallel with corpus annotation and editing a corpus-based dictionary), and we will consider requirements and feasible structures for a database that can serve both corpus and dictionary.}
}

@inproceedings{hanke:12029:sign-lang:lrec,
  author    = {Hanke, Thomas and K{\"o}nig, Susanne and Konrad, Reiner and Langer, Gabriele},
  title     = {Towards tagging of multi-sign lexemes and other multi-unit structures},
  pages     = {67--68},
  editor    = {Crasborn, Onno and Efthimiou, Eleni and Fotinea, Stavroula-Evita and Hanke, Thomas and Kristoffersen, Jette and Mesch, Johanna},
  booktitle = {Proceedings of the {LREC2012} 5th Workshop on the Representation and Processing of Sign Languages: Interactions between Corpus and Lexicon},
  maintitle = {8th International Conference on Language Resources and Evaluation ({LREC} 2012)},
  publisher = {{European Language Resources Association (ELRA)}},
  address   = {Istanbul, Turkey},
  day       = {27},
  month     = may,
  year      = {2012},
  language  = {english},
  url       = {https://www.sign-lang.uni-hamburg.de/lrec/pub/12029.html},
  abstract  = {With the building of larger sign language corpora tagging, handling and analysing large amounts of data reach a new level of complexity. Efficiency and interpersonal consistency in tagging are relevant issues as well as procedures and structures to identify and tag relevant linguistic units and structures beyond and above the manual sign level. We present and discuss problems and possible solution approaches (focussing on the working environment of iLex) of how to deal with multi-unit structures and more specifically multi-sign lexemes in annotation and lexicon building.}
}

@inproceedings{konrad:12023:sign-lang:lrec,
  author    = {Konrad, Reiner and Hanke, Thomas and K{\"o}nig, Susanne and Langer, Gabriele and Matthes, Silke and Nishio, Rie and Regen, Anja},
  title     = {From form to function. A database approach to handle lexicon building and spotting token forms in sign languages},
  pages     = {87--94},
  editor    = {Crasborn, Onno and Efthimiou, Eleni and Fotinea, Stavroula-Evita and Hanke, Thomas and Kristoffersen, Jette and Mesch, Johanna},
  booktitle = {Proceedings of the {LREC2012} 5th Workshop on the Representation and Processing of Sign Languages: Interactions between Corpus and Lexicon},
  maintitle = {8th International Conference on Language Resources and Evaluation ({LREC} 2012)},
  publisher = {{European Language Resources Association (ELRA)}},
  address   = {Istanbul, Turkey},
  day       = {27},
  month     = may,
  year      = {2012},
  language  = {english},
  url       = {https://www.sign-lang.uni-hamburg.de/lrec/pub/12023.html},
  abstract  = {Using a database with type entries that are linked to token tags in transcripts has the advantage that consistency in lemmatising is not depending on ID-glosses. In iLex types are organised in different levels. The type hierarchy allows for analysing form, iconic value, and conventionalised meanings of a sign (sub-types). Tokens can be linked either to types or sub-types.
\par
We expanded this structure for modelling sign inflection and modification as well as phonological variation. Differences between token and type form are grouped by features, called qualifiers, and specified by feature values (vocabularies). Built-in qualifiers allow for spotting the form difference when lemmatising. This facilitates lemma revision and helps to get a clear picture of how inflection, modification, or phonological variation is distributed among lexical signs. This is also a strong indicator for further POS tagging. In the long term this approach will extend the lexical database from citation-form closer to  full-form.
\par
The paper will explain the type hierarchy and introduce the qualifiers used up-to-date. Further on the handling and how the data are displayed will be illustrated. As we report work in progress in the context of the DGS corpus project, the modelling is far from complete.}
}

@inproceedings{nishio:10026:sign-lang:lrec,
  author    = {Nishio, Rie and Hong, Sung-Eun and K{\"o}nig, Susanne and Konrad, Reiner and Langer, Gabriele and Hanke, Thomas and Rathmann, Christian},
  title     = {Elicitation methods in the {DGS} ({German} {Sign} {Language}) Corpus Project},
  pages     = {178--185},
  editor    = {Dreuw, Philippe and Efthimiou, Eleni and Hanke, Thomas and Johnston, Trevor and Mart{\'i}nez Ruiz, Gregorio and Schembri, Adam},
  booktitle = {Proceedings of the {LREC2010} 4th Workshop on the Representation and Processing of Sign Languages: Corpora and Sign Language Technologies},
  maintitle = {7th International Conference on Language Resources and Evaluation ({LREC} 2010)},
  publisher = {{European Language Resources Association (ELRA)}},
  address   = {Valletta, Malta},
  day       = {22--23},
  month     = may,
  year      = {2010},
  language  = {english},
  url       = {https://www.sign-lang.uni-hamburg.de/lrec/pub/10026.html},
  abstract  = {The DGS Corpus Project is a long-term project with two major aims: (i) to establish an extensive corpus of DGS and (ii) to develop a comprehensive dictionary of DGS-German based on the analysis of the corpus data. During the first three years the main focus is on data collection. Before setting up the corpus design we conducted a survey to get an overview on the existing elicitation materials. The design of our data collection contains a variety of different stimuli and tasks with the special attention to free conversation, dialogues and monologues. To this effect, a range of possible discourse modes were considered: narration and renarration, discussion, report and description. The stimuli include pictures, picture stories, non-verbal film clips (e.g. cartoons and realistic film clips) and signed movies. In order to minimize the influence of the surrounding spoken/written language, written German is not used if possible. Introduction and explanation of each task is provided in DGS in form of movie clips. All tasks were tested in a pilot phase to examine their feasibility and reliability. Some of the tasks tested needed to go through several rounds of modifications while others did not work at all and thus were excluded from the data collection. In this paper, we not only present the tasks for elicitation and stimuli, but also describe their development process. We also discuss reasons why some stimuli were adopted from other projects while others had to be developed specifically for the purpose of our project.}
}

@inproceedings{konig:08017:sign-lang:lrec,
  author    = {K{\"o}nig, Lutz and K{\"o}nig, Susanne and Konrad, Reiner and Langer, Gabriele},
  title     = {Corpus-based Sign Dictionaries of Technical Terms -- Dictionary Projects at the {IDGS} in {Hamburg}},
  pages     = {94--100},
  editor    = {Crasborn, Onno and Efthimiou, Eleni and Hanke, Thomas and Thoutenhoofd, Ernst D. and Zwitserlood, Inge},
  booktitle = {Proceedings of the {LREC2008} 3rd Workshop on the Representation and Processing of Sign Languages: Construction and Exploitation of Sign Language Corpora},
  maintitle = {6th International Conference on Language Resources and Evaluation ({LREC} 2008)},
  publisher = {{European Language Resources Association (ELRA)}},
  address   = {Marrakech, Morocco},
  day       = {1},
  month     = jun,
  year      = {2008},
  language  = {english},
  url       = {https://www.sign-lang.uni-hamburg.de/lrec/pub/08017.html},
  abstract  = {At the Institute of German Sign Language (IDGS), six dictionary projects in such diverse technical fields as computer technology, psychology, joinery, domestic science, social work as well as health and nursing care have been carried out. A seventh project on landscaping and horticulture is in progress. Six of the seven dictionaries are based on a corpus collected from deaf experts in the respective fields. Elicitation methods, such as interviews and picture prompts, corpus design as well as annotation, transcription, sign analysis and dictionary production have been continually developed and refined over the years. Many procedures rely heavily on the use of a relational database system iLex (see other presentation).
\par
The presentation provides an overview of the projects, procedures and products with special attention given to the issues of corpus-building for and corpus-relatedness of the dictionaries at most stages of analysis and production. We focus on the corpus-based selection process which translations to include in the dictionary and on the analysis of single signs.
\par
From 1998 on, the dictionaries do not only provide translations of technical terms into DGS but also include a special section that lists single signs used in these translations in separate entries. The structure of these entries is similar to what you would expect from a general sign language dictionary. Information including lexical status, meaning, use of space, iconic value and cross references to similar signs is given for each sign. However, due to the limited size of each of these corpora and the elicitation methods used, not all information can be drawn from or validated by the corpus.
\par
Within the scope of the projects, assumptions and practical decisions have been made to deal with lexicological and lexicographical issues. These include the identification of lexemes, the degree of lexicalisation, i.e. the lexical status of signs and their meanings, the role of mouthings, and the relations between signs (polysemes vs. homonyms, modifications and variants). One important criterion for these decisions is the iconic value of signs.
\par
The lexicographic solutions applied to specialised sign language dictionaries also provide a solid basis for general sign lexicography as well as corpus annotation and lexical analysis.}
}

@inproceedings{prillwitz:08018:sign-lang:lrec,
  author    = {Prillwitz, Siegmund and Hanke, Thomas and K{\"o}nig, Susanne and Konrad, Reiner and Langer, Gabriele and Schwarz, Arvid},
  title     = {{DGS} {Corpus} Project -- Development of a Corpus Based Electronic Dictionary {German} {Sign} {Language} / {German}},
  pages     = {159--164},
  editor    = {Crasborn, Onno and Efthimiou, Eleni and Hanke, Thomas and Thoutenhoofd, Ernst D. and Zwitserlood, Inge},
  booktitle = {Proceedings of the {LREC2008} 3rd Workshop on the Representation and Processing of Sign Languages: Construction and Exploitation of Sign Language Corpora},
  maintitle = {6th International Conference on Language Resources and Evaluation ({LREC} 2008)},
  publisher = {{European Language Resources Association (ELRA)}},
  address   = {Marrakech, Morocco},
  day       = {1},
  month     = jun,
  year      = {2008},
  language  = {english},
  url       = {https://www.sign-lang.uni-hamburg.de/lrec/pub/08018.html},
  abstract  = {The poster introduces a 15-year project accepted for funding by the Hamburg Academy of Sciences. The proposed project aims to combine the collection of a large corpus with the development and production of a comprehensive, corpus based electronic dictionary of German Sign Language (DGS).
\par
To this aim, a corpus of approximately 350--400 hours from 250--300 informants will be collected in a variety of elicitation settings. This is, in size and scope, comparable to large spoken language corpora. The design allows the use of the corpus for various tasks. These are, amongst others: (i) the validation by corpus data of a basic vocabulary compiled from different published sources; (ii) research on DGS grammar based on detailed transcription data; (iii) identification of different meanings and collocations of a sign by appropriate contexts. Furthermore, the design anticipates a comparative sociolinguistic study comparable in kind and quality to Lucas et al. (2001) and Schembri/Johnston (2004). The corpus thus provides a starting point for research deep into the structure and lexicon of German Sign Language as well as into the visual-gestural mode of sign languages in general. Parts of the annotated corpus, i.e. transcription files with English translations, will be made available online to the international linguistic community.
\par
The corpus data will undergo two stages of transcription. First, a basic transcription serves to segment utterances and to identify lexical items and thus provides a first access to the data. Second, approximately 50 {\%} of the transcriptions will be transcribed again in more detail. This serves the purpose of clarifying grammatical questions for the dictionary grammar as well as dealing with lexicological and lexicographic issues. The annotation of the corpus will be closely intertwined with the requirements of lexical analysis. A high quality of transcription will be achieved through continuous verification by native signers. A relational database (iLex, cf. Hanke/Storz) supports this process, especially the consistency of type- token matching.
\par
Lexical analysis and lexicographic decisions concerning for example lexical status, language change, and lemma selection will be continuously validated by a deaf focus group and a general voting web interface which will be open for all interested members of the deaf community.
\par
The dictionary will be entirely based on the corpus with respect to the list of lemmas to be included but decidedly exceed a conglomeration of corpus references. Rather, we will systematically abstract from the references to obtain a generalized description of lexical items. Examples of sign uses will be taken directly from the corpus.
\par
For cross-linguistic research and comparability of results across projects, we consider it essential to push standardisation or at least compatibility of annotation and transcription conventions. To reach this, we have arranged cooperations with some other national corpus projects and look forward to cooperate with more projects currently in preparation.
\par
References
\par
Lucas, Ceil / Bayley, Robert / Valli, Clayton (2001): Sociolinguistic Variation in American Sign Language. Washington, DC: Gallaudet Univ. Press.
\par
Schembri, Adam / Johnston, Trevor (2004): Sociolinguistic variation in Auslan (Australian Sign Language). A research project in progress. In: Deaf Worlds 20 (1), 78-90.}
}

