@InProceedings{shiokawa-EtAl:2018:NLPTEA,
  author    = {Shiokawa, Hayato  and  Kawaguchi, Kota  and  Han, Bingcai  and  Utsuro, Takehito  and  Kawada, Yasuhide  and  Yoshioka, Masaharu  and  Kando, Noriko},
  title     = {Measuring Beginner Friendliness of Japanese Web Pages explaining Academic Concepts by Integrating Neural Image Feature and Text Features},
  booktitle = {Proceedings of the 5th Workshop on Natural Language Processing Techniques for Educational Applications},
  month     = {July},
  year      = {2018},
  address   = {Melbourne, Australia},
  publisher = {Association for Computational Linguistics},
  pages     = {143--151},
  abstract  = {Search engine is an important tool of modern academic study, but the results are lack of measurement of beginner friendliness. In order to improve the efficiency of using search engine for academic study, it is necessary to invent a technique of measuring the beginner friendliness of a Web page explaining academic concepts and to build an automatic measurement system. This paper studies how to integrate heterogeneous features such as a neural image feature generated from the image of the Web page by a variant of CNN (convolutional neural network) as well as text features extracted from the body text of the HTML file of the Web page. Integration is performed through the framework of the SVM classifier learning. Evaluation results show that heterogeneous features perform better than each individual type of features.},
  url       = {http://www.aclweb.org/anthology/W18-3721}
}

