@phdthesis {pub3076,
	title = {Towards natural speech acquisition: incremental word learning with limited data},
	author = {Irene Ayll{\'o}n Clemente},
	year = {2015},
	month = {October},
	abstract = {A strong trend in robotics is the investigation of adaptable machine learning algorithms
and frameworks that enhance the skills and application of artificial systems during the
interaction  with  humans.   The  use  of  language  is  one  of  the  most  convenient  human
methods to communicate with artificial agents.  In the last decades, the introduction of
automatic speech recognition (ASR) systems achieved important advances in the field,
however the simple and natural style of how parents teach speech to their children is still
an open question under investigation.  With the goal of improving the interaction and
learning process with artificial agents,  these must be able to increase their vocabulary
(acquire novel terms) in a satisfying manner (rapid and adequate) for the user/tutor.
In this work, we introduce an incremental word learning system to enhance speech acquisition  in  artificial  agents.   Here,  different  word-models  are  successively  learned  in  a
framework  that  possesses  little  prior  knowledge  and  is  inspired  by  the  human  infants
language acquisition process.  In order to build a user-friendly system that requires a low
tutoring time our approaches are built to cope with a small number of training samples.
Relative to the used architecture, we employ a hidden Markov model (HMM) framework
similar to most ASR systems.  Although HMMs are a powerful tool, for obtaining good
recognition  scores,  special  attention  is  required  for  the  quantity  of  samples  employed,
the bootstrapping method and the performance of the discriminative training techniques
integrated in the framework.  Therefore, we present several procedures to overcome these
challenges.   A  main  drawback  of  employing  few  training  data  samples  is  the  overspecialization of the learned models, which complicates the recognition of unseen items.  In
this context, we propose a novel computation of a parameter, which is adapted accord-
ing to the amount of provided data samples, and analyze different influences for limited
learning data.  Afterwards, we describe the proposed initialization technique introduced
in the system to properly bootstrap the estimates of the newly created model.  In this
approach, we combine unsupervised and supervised training methods with the following
re-building of the model through a multiple sequence alignment method, which arranges
and incorporates the succession of hidden states obtained by the Viterbi decoding algorithm.  Next, several large margin (LM) discriminative training methods are analyzed to
increase the generalization performance of the previously created models, i.e.  improving
the classification of the models.  Here,  we propose different procedures appropriate for
employing  limited  data  in  discriminative  training.   Finally,  the  proposed  methods  are
compared against state-of-the-art techniques during the experimental phase.  Similarly,
each individual contribution of the introduced approaches is measured in relation to the
global  yielded  improvement  of  the  whole  framework.   Additionally,  we  examine  a  potential  decrease  of  the  amount  of  training  samples  with  the  purpose  of  decreasing  the
time employed to teach a robot.  The evaluation of our approaches is realized on differ-
ent recognition tasks containing isolated and continuous digits.  After all, we demonstrate
that the introduced collection of techniques achieve important improvements so that they
represent a significant step towards efficient incremental word learning with limited data.},
	publisher = {University of Bielefeld},
	booktitle = {University of Bielefeld}
}
