@inproceedings{jui:22018:sign-lang:lrec,
  author    = {Jui, Tonni Das and Bejarano, Gissella and Rivas, Pablo},
  title     = {A Machine Learning-based Segmentation Approach for Measuring Similarity between Sign Languages},
  pages     = {94--101},
  editor    = {Efthimiou, Eleni and Fotinea, Stavroula-Evita and Hanke, Thomas and Hochgesang, Julie A. and Kristoffersen, Jette and Mesch, Johanna and Schulder, Marc},
  booktitle = {Proceedings of the {LREC2022} 10th Workshop on the Representation and Processing of Sign Languages: Multilingual Sign Language Resources},
  maintitle = {13th International Conference on Language Resources and Evaluation ({LREC} 2022)},
  publisher = {{European Language Resources Association (ELRA)}},
  address   = {Marseille, France},
  day       = {25},
  month     = jun,
  year      = {2022},
  isbn      = {979-10-95546-86-3},
  language  = {english},
  url       = {https://www.sign-lang.uni-hamburg.de/lrec/pub/22018.html},
  abstract  = {Due to the lack of more variate, native and continuous datasets, sign languages are low-resources languages that can benefit from multilingualism in machine translation. In order to analyze the benefits of approaches like multilingualism, finding the similarity between sign languages can guide better matches and contributions between languages. However, calculating the similarity between sign languages again implies a laborious work to measure how close or distant signs are and their respective contexts. For that reason, we propose to support the similarity measurement between sign languages through a video-segmentation-based machine learning model that will quantify this match among signs of different countries' sign languages. Using a machine learning approach the similarity measurement process can run more smoothly, compared to a more manual approach. We use a pre-trained temporal segmentation model for British Sign Language (BSL). We test it on three datasets, an American Sign Language (ASL) dataset, an Indian Sign Language (ISL), and an Australian Sign Language (AUSLAN) dataset. We hypothesize that the percentage of segmented and recognized signs by this machine learning model can represent the percentage of overlap or similarity between British and the other three sign languages. In our ongoing work, we evaluate three metrics considering Swadesh's and Woodward's list and their synonyms. We found that our intermediate-strict metric coincides with a more classical analysis of the similarity between British and American Sign Language, as well as with the classical low measurement between Indian and British sign languages. On the other hand, our similarity measurement between British and Australian Sign language just holds for part of the Australian Sign Language and not the whole data sample.}
}

@inproceedings{smith:22017:sign-lang:lrec,
  author    = {Smith, River Tae and Willoughby, Louisa and Johnston, Trevor},
  title     = {Integrating {Auslan} Resources into the Language Data Commons of {Australia}},
  pages     = {181--186},
  editor    = {Efthimiou, Eleni and Fotinea, Stavroula-Evita and Hanke, Thomas and Hochgesang, Julie A. and Kristoffersen, Jette and Mesch, Johanna and Schulder, Marc},
  booktitle = {Proceedings of the {LREC2022} 10th Workshop on the Representation and Processing of Sign Languages: Multilingual Sign Language Resources},
  maintitle = {13th International Conference on Language Resources and Evaluation ({LREC} 2022)},
  publisher = {{European Language Resources Association (ELRA)}},
  address   = {Marseille, France},
  day       = {25},
  month     = jun,
  year      = {2022},
  isbn      = {979-10-95546-86-3},
  language  = {english},
  url       = {https://www.sign-lang.uni-hamburg.de/lrec/pub/22017.html},
  abstract  = {This paper describes a project to secure Auslan (Australian Sign Language) resources within a national language data network called the Language Data Commons of Australia (LDaCA). The resources are Auslan Signbank, a web-based multi-media dictionary, and the Auslan Corpus, a collection of video recordings of the language being used in various contexts with time-aligned ELAN annotation files. We aim to make these resources accessible to the language community, encourage community participation in the curation of the data, and facilitate and extend their uses in language teaching and linguistic research. The software platforms of both resources will be made compatible with other LDaCA resources; and the two will also be aggregated and linked so that (i) users of the dictionary can view attested corpus examples for an entry; and (ii) users of the corpus can instantly view the dictionary entry for an already glossed sign to check phonological, lexical and grammatical information about it, and/or to ensure that the correct annotation gloss (aka `ID-gloss') for a sign token has been chosen. This will enhance additions to annotations in the Auslan Corpus, entries in Auslan Signbank and the integrity of research based on both.}
}

@inproceedings{kuder:18033:sign-lang:lrec,
  author    = {Kuder, Anna and Filipczak, Joanna and Mostowski, Piotr and Rutkowski, Pawe{\l} and Johnston, Trevor},
  title     = {What Corpus-based Research on Negation in {Auslan} and {PJM} Tells Us about Building and Using Sign Language Corpora},
  pages     = {101--106},
  editor    = {Bono, Mayumi and Efthimiou, Eleni and Fotinea, Stavroula-Evita and Hanke, Thomas and Hochgesang, Julie A. and Kristoffersen, Jette and Mesch, Johanna and Osugi, Yutaka},
  booktitle = {Proceedings of the {LREC2018} 8th Workshop on the Representation and Processing of Sign Languages: Involving the Language Community},
  maintitle = {11th International Conference on Language Resources and Evaluation ({LREC} 2018)},
  publisher = {{European Language Resources Association (ELRA)}},
  address   = {Miyazaki, Japan},
  day       = {12},
  month     = may,
  year      = {2018},
  isbn      = {979-10-95546-01-6},
  language  = {english},
  url       = {https://www.sign-lang.uni-hamburg.de/lrec/pub/18033.html},
  abstract  = {In this paper, we would like to discuss our current work on negation in Auslan (Australian Sign Language) and PJM (Polish Sign Language, polski j{\k e}zyk migowy) as an example of experience in using sign language corpus data for research purposes. We describe how we prepared the data for two detailed empirical studies, given similarities and differences between the Australian and Polish corpus projects. We present our findings on negation in both languages, which turn out to be surprisingly similar. At the same time, what the two corpus studies show seems to be quite different from many previous descriptions of sign language negation found in the literature. Some remarks on how to effectively plan and carry out the annotation process of sign language texts are outlined at the end of the present paper, as they might be helpful to other researchers working on designing a corpus. Our work leads to two main conclusions: (1) in many cases, usage data may not be easily reconciled with intuitions and assumptions about how sign languages function and what their grammatical characteristics are like, (2) in order to obtain representative and reliable data from large-scale corpora one needs to plan and carry out the annotation process very thoroughly.}
}

@inproceedings{cassidy-etal-2018-signbank:lrec,
  author    = {Cassidy, Steve and Crasborn, Onno and Nieminen, Henri and Stoop, Wessel and Hulsbosch, Micha and Even, Susan and Komen, Erwin and Johnston, Trevor},
  title     = {{S}ignbank: Software to Support Web Based Dictionaries of Sign Language},
  pages     = {2359--2364},
  editor    = {Calzolari, Nicoletta and Choukri, Khalid and Cieri, Christopher and Declerck, Thierry and Goggi, Sara and Hasida, Koiti and Isahara, Hitoshi and Maegaard, Bente and Mariani, Joseph and Mazo,  H{\'e}l{\`e}ne and Moreno, Asuncion and Odijk, Jan and Piperidis, Stelios and Tokunaga, Takenobu},
  booktitle = {11th International Conference on Language Resources and Evaluation ({LREC} 2018)},
  publisher = {{European Language Resources Association (ELRA)}},
  address   = {Miyazaki, Japan},
  day       = {7--12},
  month     = may,
  year      = {2018},
  isbn      = {979-10-95546-00-9},
  language  = {english},
  url       = {https://aclanthology.org/L18-1374},
  abstract  = {Signbank is a web application that was originally built to support the Auslan Signbank on-line web dictionary, it was an Open Source re-implementation of an earlier version of that site. The application provides a framework for the development of a rich lexical database of sign language augmented with video samples of signs. As an Open Source project, the original Signbank has formed the basis of a number of new sign language dictionaries and corpora including those for British Sign Language, Sign Language of the Netherlands and Finnish Sign Language. Versions are under development for American Sign Language and Flemish Sign Language. This paper describes the overall architecture of the Signbank system and its representation of lexical entries and associated entities.}
}

@inproceedings{johnston:14003:sign-lang:lrec,
  author    = {Johnston, Trevor and van Roekel, Jane},
  title     = {Mouth-based non-manual coding schema used in the {Auslan} corpus: Explanation, application and preliminary results},
  pages     = {81--88},
  editor    = {Crasborn, Onno and Efthimiou, Eleni and Fotinea, Stavroula-Evita and Hanke, Thomas and Hochgesang, Julie A. and Kristoffersen, Jette and Mesch, Johanna},
  booktitle = {Proceedings of the {LREC2014} 6th Workshop on the Representation and Processing of Sign Languages: Beyond the Manual Channel},
  maintitle = {9th International Conference on Language Resources and Evaluation ({LREC} 2014)},
  publisher = {{European Language Resources Association (ELRA)}},
  address   = {Reykjavik, Iceland},
  day       = {31},
  month     = may,
  year      = {2014},
  language  = {english},
  url       = {https://www.sign-lang.uni-hamburg.de/lrec/pub/14003.html},
  abstract  = {We describe a corpus-based study of one type of non-manual in signed languages (SLs) --- mouth actions. Our ultimate aim is to examine the distribution and characteristics of mouth actions in Auslan (Australian Sign Language) to gauge the degree of language-specific conventionalization of these forms. We divide mouth gestures into categories broadly based on Crasborn et al. (2008), but modified to accommodate our experiences with the Auslan data. All signs and all mouth actions are examined and the state of the mouth in each sign is assigned to one of three broad categories: (i) mouthings, (ii) mouth gestures, and (iii) no mouth action. Mouth actions that invariably occur while communicating in SLs have posed a number of questions for linguists: which are `merely borrowings' from the relevant ambient spoken language (SpL)? which are gestural and shared with all of the members of the wider community in which signers find themselves? and which are conventionalized aspects of the grammar of some or all SLs? We believe these schema captures all the relevant information about mouth forms and their use and meaning in context to enable us to describe their function and degree of conventionality.}
}

@inproceedings{cormier:12033:sign-lang:lrec,
  author    = {Cormier, Kearsy and Fenlon, Jordan and Johnston, Trevor and Rentelis, Ramas and Schembri, Adam and Rowley, Katherine and Adam, Robert and Woll, Bencie},
  title     = {From Corpus to Lexical Database to Online Dictionary: Issues in annotation of the {BSL} Corpus and the Development of {BSL} {SignBank}},
  pages     = {7--12},
  editor    = {Crasborn, Onno and Efthimiou, Eleni and Fotinea, Stavroula-Evita and Hanke, Thomas and Kristoffersen, Jette and Mesch, Johanna},
  booktitle = {Proceedings of the {LREC2012} 5th Workshop on the Representation and Processing of Sign Languages: Interactions between Corpus and Lexicon},
  maintitle = {8th International Conference on Language Resources and Evaluation ({LREC} 2012)},
  publisher = {{European Language Resources Association (ELRA)}},
  address   = {Istanbul, Turkey},
  day       = {27},
  month     = may,
  year      = {2012},
  language  = {english},
  url       = {https://www.sign-lang.uni-hamburg.de/lrec/pub/12033.html},
  abstract  = {One requirement of a sign language corpus is that it should be machine-readable, but only a systematic approach to annotation that involves lemmatisation of the sign language glosses can make this possible at the present time. Such lemmatisation involves grouping morphological and phonological variants together into a single lemma, so that all related variants of a sign can be identified and analysed as a single sign. This lemmatisation process is made more straightforward by the existence of a comprehensive lexical database, as in the case for Australian Sign Language (Auslan). When annotation of data collected as part of the British Sign Language (BSL) Corpus Project began, no such lexical database for BSL existed. Therefore, a lemmatised BSL lexical database was created concurrently during annotation of the BSL Corpus data. As part of ongoing work by the Deafness Cognition {\&} Language Research Centre, this lexical database is being developed into an online BSL dictionary, BSL SignBank. This paper describes the adaptation of the Auslan lexical database into a BSL lexical database, and the current development of this lexical database into BSL SignBank.}
}

@inproceedings{johnston:10002:sign-lang:lrec,
  author    = {Johnston, Trevor},
  title     = {Adding value to, and extracting of value from, a signed language corpus through secondary processing: implications for annotation schemas and corpus creation},
  pages     = {137--142},
  editor    = {Dreuw, Philippe and Efthimiou, Eleni and Hanke, Thomas and Johnston, Trevor and Mart{\'i}nez Ruiz, Gregorio and Schembri, Adam},
  booktitle = {Proceedings of the {LREC2010} 4th Workshop on the Representation and Processing of Sign Languages: Corpora and Sign Language Technologies},
  maintitle = {7th International Conference on Language Resources and Evaluation ({LREC} 2010)},
  publisher = {{European Language Resources Association (ELRA)}},
  address   = {Valletta, Malta},
  day       = {22--23},
  month     = may,
  year      = {2010},
  language  = {english},
  url       = {https://www.sign-lang.uni-hamburg.de/lrec/pub/10002.html},
  abstract  = {A basic signed language (SL) corpus is created through primary processing of video recordings using multi{\_}media annotation software. Primary processing entails the tokenization and identification of SL units. For the purposes of linguistic research a corpus also needs secondary processing. Secondary processing entails appending tags for specific linguistic features to primary annotations. I draw on the experience from the Auslan corpus project to describe how primary and secondary processing can be used in corpus-based SL research. In particular, I show how the tier structure of ELAN can be used to tag SL units in a variety of ways, and how this information can be used to glean new information from the corpus which can then be added as new annotations to the corpus. Value-adding by principled and systematic primary and secondary processing of digital recordings is thus not only essential for corpus creation ('machine-readability'), it also enables further enriching of the corpus so that even more value can be extracted. I conclude by discussing the implications for annotation software and standardized annotation schemas used in the creation of SL corpora.}
}

@inproceedings{debeuzeville:08020:sign-lang:lrec,
  author    = {de Beuzeville, Louise},
  title     = {Pointing and verb modification: the expression of semantic roles in the {Auslan} Corpus},
  pages     = {13--16},
  editor    = {Crasborn, Onno and Efthimiou, Eleni and Hanke, Thomas and Thoutenhoofd, Ernst D. and Zwitserlood, Inge},
  booktitle = {Proceedings of the {LREC2008} 3rd Workshop on the Representation and Processing of Sign Languages: Construction and Exploitation of Sign Language Corpora},
  maintitle = {6th International Conference on Language Resources and Evaluation ({LREC} 2008)},
  publisher = {{European Language Resources Association (ELRA)}},
  address   = {Marrakech, Morocco},
  day       = {1},
  month     = jun,
  year      = {2008},
  language  = {english},
  url       = {https://www.sign-lang.uni-hamburg.de/lrec/pub/08020.html},
  abstract  = {As part of a larger project investigating the grammatical use of space in Auslan, 62 texts from the Auslan Corpus were annotated and analysed for the spatial modification of verbs to show semantic roles. Data were taken from two groups of native and near-native Auslan signers. Spontaneous narratives were sourced from a sociolinguistic variation corpus collected from 211 participants all over Australia. The second set of texts was elicited from 100 adult native signers of Auslan from the Auslan Corpus Project. Participants retold to a deaf interlocutor a prepared Aesop's fable and a spontaneous personal recount of a memorable event, as well as answering a series of questions on their attitudes to various factors influencing the deaf community (such as genetic testing and cochlear implants). The texts from both corpora were recorded on digital videotape and then annotated using ELAN software. Here we report on 62 texts that have been annotated (approximately 9,000 signs from 50 narrative texts and 9,000 from 10 attitude surveys). Each sign or meaningful gesture was identified, with points being categorised as pronouns or other. These signs were then classified into word class and the nouns and verbs tagged for whether they could be modified spatially. Next, the indicating nouns and verbs were annotated as to whether or not their spatial modification was realised. In this paper, we discuss the use of the ELAN search functions across multiple files in order to identify the proportion of sign types in the texts and the frequency with which indicating verbs are actually modified for space. We then searched all files again to identify all instances where pointing signs occurred directly before or after an indicating verb, in order to calculate whether the collocation of a point (pronoun, other or either) and a non-modified indicating verb was statistically significant. Despite the claim that indicating verbs in signed languages are obligatorily modified (`inflected') with respect to loci in the signing space in order to show person `agreement', we found that these verbs are actually only spatially modified about on third of the time (Johnston et al and de B et al., forthcoming) and this study showed that to be partly as a result of presence of points. The results help determine where and when the spatial modification of indicating verbs is used in natural Auslan texts (and potentially other signed languages). Based on this data, we suggest that 1) the degree of grammaticalization of indicating verbs may not be as great as once thought and 2) the apparent non-obligatory or variable use of spatial modifications may be partly accounted for by the presence of pointing signs---very frequent in signed texts---before or directly after the verb.
\par
References
\par
Johnston, T. A., de Beuzeville, L., Schembri, A., {\&} Goswell, D. (2007) On not missing the point: indicating verbs in Auslan. Paper presented at the 10th International Cognitive Linguistics Conference, Krakow, Poland, July 15th -- 20th, 2007
\par
de Beuzeville, L., Johnston, T. A., Schembri, A., {\&} Goswell, D. (forthcoming) The use of space with lexical verbs in Auslan.}
}

@inproceedings{johnston:08031:sign-lang:lrec,
  author    = {Johnston, Trevor},
  title     = {Corpus linguistics and signed languages: no lemmata, no corpus},
  pages     = {82--87},
  editor    = {Crasborn, Onno and Efthimiou, Eleni and Hanke, Thomas and Thoutenhoofd, Ernst D. and Zwitserlood, Inge},
  booktitle = {Proceedings of the {LREC2008} 3rd Workshop on the Representation and Processing of Sign Languages: Construction and Exploitation of Sign Language Corpora},
  maintitle = {6th International Conference on Language Resources and Evaluation ({LREC} 2008)},
  publisher = {{European Language Resources Association (ELRA)}},
  address   = {Marrakech, Morocco},
  day       = {1},
  month     = jun,
  year      = {2008},
  language  = {english},
  url       = {https://www.sign-lang.uni-hamburg.de/lrec/pub/08031.html},
  abstract  = {A fundamental problem in the creation of signed language corpora is lemmatisation. Lemmatisation---the classification or identification of related word forms under a single label or lemma (the equivalent of headwords or headsigns in a dictionary)---is central to the process of corpus creation. The reason is that signed language corpora---as with all modern linguistic corpora---need to be machine-readable and this means that sign annotations should not only be informed by linguistic theory but also that tags appended to these annotations should be used consistently and systematically. In addition, a corpus must also be well documented (i.e., with accurate and relevant metadata) and representative of the language community (i.e., of relevant registers and sociolinguistic). All this requires dedicated technology (e.g., ELAN), standards and protocols (e.g., IMDI metadata descriptors), and transparent and agreed grammatical tags (e.g., grammatical class labels). However, it also requires the identification of lemmata and this presupposes the unique identification of sign forms. In other words, a successful corpus project presupposes the availability of a reference dictionary or lexical database to facilitate lemma identification and consistency in lemmatisation. Without lemmatisation a collection of recordings with various related appended annotation files will not be able to be used as a true linguistic corpus as the counting, sorting, tagging. etc. of types and tokens is rendered virtually impossible. This presentation draws on the Australian experience of corpus creation to show how a dictionary in the form of a computerized lexical database needs to be created and integrated into any signed language corpus project. Plans for the creation of new signed language corpora will be seriously flawed if they do not take this into account.}
}

