@inproceedings{jahn:18018:sign-lang:lrec,
  author    = {Jahn, Elena and Konrad, Reiner and Langer, Gabriele and Wagner, Sven and Hanke, Thomas},
  title     = {Publishing {DGS} {Corpus} Data: Different Formats for Different Needs},
  pages     = {83--90},
  editor    = {Bono, Mayumi and Efthimiou, Eleni and Fotinea, Stavroula-Evita and Hanke, Thomas and Hochgesang, Julie A. and Kristoffersen, Jette and Mesch, Johanna and Osugi, Yutaka},
  booktitle = {Proceedings of the {LREC2018} 8th Workshop on the Representation and Processing of Sign Languages: Involving the Language Community},
  maintitle = {11th International Conference on Language Resources and Evaluation ({LREC} 2018)},
  publisher = {{European Language Resources Association (ELRA)}},
  address   = {Miyazaki, Japan},
  day       = {12},
  month     = may,
  year      = {2018},
  isbn      = {979-10-95546-01-6},
  language  = {english},
  url       = {https://www.sign-lang.uni-hamburg.de/lrec/pub/18018.html},
  abstract  = {In 2010-2012, the DGS-Korpus project collected a large corpus of German Sign Language (DGS). Now, a substantial subset of the data is published, namely the Public DGS Corpus. We describe the considerations and decisions taken regarding what part of the data is to be made public, the necessary quality assurance measures to the data preparation as well as the formats of the published data. The corpus is published in three different ways in order to fulfil the needs of a variety of different users. First of all, the data is made available to the language community whose members allowed us to share their recorded language. In addition, we hope that a large number of non-scientific users with various backgrounds will find the data useful. Last but not least, we aim to make the data attractive for users with a scientific background and provide the possibility to conduct studies based on it, irrespective of whether they are familiar with DGS or not.}
}

@inproceedings{hanke:10047:sign-lang:lrec,
  author    = {Hanke, Thomas and K{\"o}nig, Lutz and Wagner, Sven and Matthes, Silke},
  title     = {{DGS} {Corpus} {\&} {Dicta-Sign}: The {Hamburg} Studio Setup},
  pages     = {106--109},
  editor    = {Dreuw, Philippe and Efthimiou, Eleni and Hanke, Thomas and Johnston, Trevor and Mart{\'i}nez Ruiz, Gregorio and Schembri, Adam},
  booktitle = {Proceedings of the {LREC2010} 4th Workshop on the Representation and Processing of Sign Languages: Corpora and Sign Language Technologies},
  maintitle = {7th International Conference on Language Resources and Evaluation ({LREC} 2010)},
  publisher = {{European Language Resources Association (ELRA)}},
  address   = {Valletta, Malta},
  day       = {22--23},
  month     = may,
  year      = {2010},
  language  = {english},
  url       = {https://www.sign-lang.uni-hamburg.de/lrec/pub/10047.html},
  abstract  = {Not taking into account budget restrictions, the setup of a sign language studio always is a balancing act between high quality recordings on the one hand not to make the transcription process even more complicated than it is anyway and possibly to enable automatic processing of the recordings, and on the other hand an environment where the informants still feel comfortable enough so that the recording situation does not have too much impact on the signing. In the case of the DGS Corpus project, an additional constraint is that the studio is to be relocated twelve times over the course of two years as it was decided to make the recordings in the regions instead of inviting participants to one central place to avoid dialectal mixing. One of the implications of this approach is that the studio is operated by non-specialist deaf fieldworkers with limited time available for training.
\par
Basically all tasks in the project involve two informants interacting in different ways with each other. A moderator (the fieldworker from the region) introduces the tasks and observes the conversation, but only interferes with the conversation if absolutely necessary.
\par
The camera setup we finally ended up with consists of seven cameras altogether, three on each informant and one for the whole scene including the moderator. Two HD cameras on the informant provide frontal and birds-eye views while a stereo camera mounted on top of the frontal-view camera provides footage that helps automatic processing. The seventh camera is an HD camera as well.
\par
In contrary to setups in earlier projects, we invite the two informants to sit down directly facing each other, with the frontal-view camera positioned above (and behind) the head of the other informant. Pre-tests revealed that with a distance of approximately three meters, the distorsion introduced by the elevated position of the camera does not negatively affect the transcription from video. Instead, this setting provides a front view of the informant similar to the addressee's, allowing to identify body shifts as well as direction of eye gaze more easily. At the same time, this constellation avoids informants targeting their signing back and forth between the addressee and the camera.
\par
Elicitation material and instructions are presented to the informants on screens located on the floor between them. A custom software, ``Session Director'' allows the moderator to present slides to the informants by the click of a button, and to keep track of the time elapsed for each individual task as well as the whole session. Using pre-recorded instructions and elicitation materials not only reduces the burdens on the moderator, but also makes sure that all informants get exactly the same input.
\par
Session Director keeps a log of all actions started by the moderator, allowing us to exactly reconstruct what task has been worked on when. This log is easily converted into tagging in our transcription environment, iLex. This not only allows automatic segmentation of tasks and pauses, but also introduces links from the transcript to the task and vice versa.
\par
Task descriptions for Session Director are kept as XML files, making it easy to use this freely available tool for other projects as well.}
}

@inproceedings{hanke:10056:sign-lang:lrec,
  author    = {Hanke, Thomas and Storz, Jakob and Wagner, Sven},
  title     = {{iLex}: Handling Multi-Camera Recordings},
  pages     = {110--111},
  editor    = {Dreuw, Philippe and Efthimiou, Eleni and Hanke, Thomas and Johnston, Trevor and Mart{\'i}nez Ruiz, Gregorio and Schembri, Adam},
  booktitle = {Proceedings of the {LREC2010} 4th Workshop on the Representation and Processing of Sign Languages: Corpora and Sign Language Technologies},
  maintitle = {7th International Conference on Language Resources and Evaluation ({LREC} 2010)},
  publisher = {{European Language Resources Association (ELRA)}},
  address   = {Valletta, Malta},
  day       = {22--23},
  month     = may,
  year      = {2010},
  language  = {english},
  url       = {https://www.sign-lang.uni-hamburg.de/lrec/pub/10056.html},
  abstract  = {Until recently, sign language researchers were quite happy with just one or two views for each recording session. While ELAN allows the user to relate several media files to a transcript and to sync them, iLex just allows one single media container and relies on the container format, such as QuickTime, to group and sync several video streams into one container. In order to save screen real estate, iLex offers the user the possibility to switch on or off individual tracks within the media file. This works quite fine with two or three different views grouped, but fails to provide an adequate solution in multi-view projects such as Dicta-Sign or DGS Corpus with seven cameras altogether for a pair of informants. The advent of HD videos makes screen real estate really an issue: Even on very large screens, video competes with transcription space.
\par
Here we present a user interface study that allows flexible switching between video layouts whenever the transcription focus changes. Switching (including zooming and cropping) may be initiated at any point of time by the user, or can be automated to depend on tagging such as tasks or turns. This user interface is backed up by a server infrastructure providing videos in different spatial resolutions as needed for optimal display while saving transfer bandwidth and local processing power which even nowadays becomes an issue when dealing with several HD videos in parallel.}
}

@inproceedings{bleicken-etal-2016-using:lrec,
  author    = {Bleicken, Julian and Hanke, Thomas and Salden, Uta and Wagner, Sven},
  title     = {Using a Language Technology Infrastructure for {G}erman in order to Anonymize {G}erman {S}ign {L}anguage Corpus Data},
  pages     = {3303--3306},
  editor    = {Calzolari, Nicoletta and Choukri, Khalid and Declerck, Thierry and Goggi, Sara and Grobelnik, Marko and Maegaard, Bente and Mariani, Joseph and Mazo, H{\'e}l{\`e}ne and Moreno, Asuncion and Odijk, Jan and Piperidis, Stelios},
  booktitle = {10th International Conference on Language Resources and Evaluation ({LREC} 2016)},
  publisher = {{European Language Resources Association (ELRA)}},
  address   = {Portoro{\v z}, Slovenia},
  day       = {23--28},
  month     = may,
  year      = {2016},
  isbn      = {978-2-9517408-9-1},
  language  = {english},
  url       = {https://aclanthology.org/L16-1526},
  doi       = {10.63317/4ry2rahpi7et},
  abstract  = {For publishing sign language corpus data on the web, anonymization is crucial even if it is impossible to hide the visual appearance of the signers: In a small community, even vague references to third persons may be enough to identify those persons. In the case of the DGS Korpus (German Sign Language corpus) project, we want to publish data as a contribution to the cultural heritage of the sign language community while annotation of the data is still ongoing. This poses the question how well anonymization can be achieved given that no full linguistic analysis of the data is available. Basically, we combine analysis of all data that we have, including named entity recognition on translations into German. For this, we use the WebLicht language technology infrastructure. We report on the reliability of these methods in this special context and also illustrate how the anonymization of the video data is technically achieved in order to minimally disturb the viewer.}
}

