This article is devoted to the verification of the empirical Heaps law in European languages using Google Books Ngram corpus data. The connection between word distribution frequency and expected dependence of individual word number on text size is analysed in terms of a simple probability model of text generation. It is shown that the Heaps exponent varies significantly within characteristic time intervals of 60-100 years.
Description
[1612.09213] Verifying Heaps' law using Google Books Ngram data
%0 Generic
%1 bochkarev2016
%A Bochkarev, Vladimir V.
%A Lerner, Eduard Yu.
%A Shevlyakova, Anna V.
%D 2016
%K heaps mybook texts
%T Verifying Heaps' law using Google Books Ngram data
%U http://arxiv.org/abs/1612.09213
%X This article is devoted to the verification of the empirical Heaps law in European languages using Google Books Ngram corpus data. The connection between word distribution frequency and expected dependence of individual word number on text size is analysed in terms of a simple probability model of text generation. It is shown that the Heaps exponent varies significantly within characteristic time intervals of 60-100 years.
@misc{bochkarev2016,
abstract = {This article is devoted to the verification of the empirical Heaps law in European languages using Google Books Ngram corpus data. The connection between word distribution frequency and expected dependence of individual word number on text size is analysed in terms of a simple probability model of text generation. It is shown that the Heaps exponent varies significantly within characteristic time intervals of 60-100 years.},
added-at = {2017-12-04T17:11:38.000+0100},
author = {Bochkarev, Vladimir V. and Lerner, Eduard Yu. and Shevlyakova, Anna V.},
biburl = {https://www.bibsonomy.org/bibtex/24af1c1bfc8613fcc594731d0a67b84a7/vitelot},
description = {[1612.09213] Verifying Heaps' law using Google Books Ngram data},
interhash = {9d2290c7be7a8b285d36a34b9bdf1b3a},
intrahash = {4af1c1bfc8613fcc594731d0a67b84a7},
keywords = {heaps mybook texts},
note = {cite arxiv:1612.09213Comment: 8 pages, 6 figures},
timestamp = {2017-12-04T18:17:35.000+0100},
title = {Verifying Heaps' law using Google Books Ngram data},
url = {http://arxiv.org/abs/1612.09213},
year = 2016
}