{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,11,18]],"date-time":"2025-11-18T11:48:10Z","timestamp":1763466490805},"reference-count":37,"publisher":"Springer Science and Business Media LLC","issue":"1","license":[{"start":{"date-parts":[[2009,8,13]],"date-time":"2009-08-13T00:00:00Z","timestamp":1250121600000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2009,8,13]],"date-time":"2009-08-13T00:00:00Z","timestamp":1250121600000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":["Inf Retrieval"],"published-print":{"date-parts":[[2010,2]]},"DOI":"10.1007\/s10791-009-9107-y","type":"journal-article","created":{"date-parts":[[2009,8,12]],"date-time":"2009-08-12T15:22:09Z","timestamp":1250090529000},"page":"70-95","update-policy":"http:\/\/dx.doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":21,"title":["Estimating deep web data source size by capture\u2013recapture method"],"prefix":"10.1007","volume":"13","author":[{"given":"Jianguo","family":"Lu","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Dingding","family":"Li","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2009,8,13]]},"reference":[{"key":"9107_CR1","unstructured":"Amstrup, S. C., McDonald, T. L., & Manly, B. F. J. (2005). Handbook of capture\u2013recapture analysis. Princeton University Press."},{"key":"9107_CR2","unstructured":"Barbosa, L., & Freire, J. (2004). Siphoning hidden-web data through keyword-based interfaces. In Proceedings of SBBD, 2004."},{"key":"9107_CR3","doi-asserted-by":"crossref","unstructured":"Bar-Yossef, Z., & Gurevich, M. (2006). Random sampling from a search engine\u2019s index. In Proceedings of WWW, 2006, pp, 367\u2013376.","DOI":"10.1145\/1135777.1135833"},{"key":"9107_CR4","doi-asserted-by":"crossref","unstructured":"Bar-Yossef, Z., & Gurevich, M. (2007). Efficient search engine measurements. In Proceedings of WWW, 2007, pp. 401\u2013410.","DOI":"10.1145\/1242572.1242627"},{"key":"9107_CR5","doi-asserted-by":"crossref","unstructured":"Bergman, M. K. (2001). The deep web: Surfacing hidden value. The Journal of Electronic Publishing, 7(1).","DOI":"10.3998\/3336451.0007.104"},{"key":"9107_CR6","doi-asserted-by":"crossref","unstructured":"Bharat, K., & Broder, A. (1998). A technique for measuring the relative size and overlap of public Web search engines. In Proceedings of WWW, 1998, pp. 379\u2013388.","DOI":"10.1016\/S0169-7552(98)00127-5"},{"key":"9107_CR7","doi-asserted-by":"crossref","unstructured":"Bolshakov, I. A., & Galicia-Haro, S. N. (2003). Can we correctly estimate the total number of pages in Google for a specific language? CICLing 2003, pp. 415\u2013419.","DOI":"10.1007\/3-540-36456-0_44"},{"key":"9107_CR9","doi-asserted-by":"crossref","unstructured":"Broder, A., Fontura, M., Josifovski, V., Kumar, R., Motwani, R., Nabar, S., et\u00a0al. (2006). Estimating corpus size via queries. In CIKM\u201906, pp. 594\u2013603.","DOI":"10.1145\/1183614.1183699"},{"issue":"2","key":"9107_CR10","doi-asserted-by":"crossref","first-page":"97","DOI":"10.1145\/382979.383040","volume":"19","author":"J. Callan","year":"2001","unstructured":"Callan, J., & Connell, M. (2001). Query-based sampling of text databases. ACM Transactions on Information Systems, 19(2), 97\u2013130.","journal-title":"ACM Transactions on Information Systems"},{"key":"9107_CR11","doi-asserted-by":"crossref","unstructured":"Caverlee, J., Liu, L., & Buttler, D. (2004). Probe, cluster, and discover: Focused extraction of QA-pagelets from the deep web. In Proceedings of ICDE 2004, pp. 103\u2013114.","DOI":"10.1109\/ICDE.2004.1319988"},{"key":"9107_CR12","doi-asserted-by":"publisher","first-page":"210","DOI":"10.2307\/2290471","volume":"87","author":"A. Chao","year":"1992","unstructured":"Chao, A., & Lee, S.-M. (1992). Estimating the number of classes via sample coverage. Journal of American Statistical Association, 87, 210\u2013217.","journal-title":"Journal of American Statistical Association"},{"key":"9107_CR13","doi-asserted-by":"crossref","unstructured":"Crescenzi, V., Mecca, G., & Merialdo, P. (2001). RoadRunner: Towards automatic data extraction from large web sites. In Proceedings of VLDB 2001, pp. 109\u2013118.","DOI":"10.1145\/564691.564778"},{"issue":"3\/4","key":"9107_CR14","doi-asserted-by":"publisher","first-page":"343","DOI":"10.2307\/2333183","volume":"45","author":"J. N. Darroch","year":"1958","unstructured":"Darroch, J. N. (1958). The multiple-recapture census: I. Estimation of a closed population. Biometrika, 45(3\/4), 343\u2013359.","journal-title":"Biometrika"},{"key":"9107_CR15","doi-asserted-by":"crossref","unstructured":"Dobra, A., & Fienberg, S. (2004). How large is the World Wide Web? Web Dynamics, Springer, pp. 23\u201344.","DOI":"10.1007\/978-3-662-10874-1_2"},{"key":"9107_CR16","doi-asserted-by":"crossref","unstructured":"Gulli, A., & Signorini A. (2005). The indexable web is more than 11.5 billion pages. In Proceedings of WWW 2005, pp. 902\u2013903.","DOI":"10.1145\/1062745.1062789"},{"key":"9107_CR17","unstructured":"Haas, P. J., Naughton, J. F., Seshadri, S., & Stokes, L. (1995). Sampling-based estimation of the number of distinct values of an attribute. In Proceedings of VLDB 1995, pp. 311\u2013322."},{"key":"9107_CR18","unstructured":"Hatcher, E., & Gospodnetic, O. (2004). Lucene in action. Manning Publications."},{"issue":"1","key":"9107_CR21","doi-asserted-by":"publisher","first-page":"154","DOI":"10.2307\/3213383","volume":"16","author":"L. Holst","year":"1979","unstructured":"Holst, L. (1979). A unified approach to limit theorems for urn models. Journal of Applied Probability, 16(1), 154\u2013162.","journal-title":"Journal of Applied Probability"},{"key":"9107_CR22","doi-asserted-by":"crossref","unstructured":"Ipeirotis, P. G., Gravano, L., & Sahami, M. (2001). Probe, count, and classify: Categorizing hidden web databases. In Proceedings of SIGMOD\u201901.","DOI":"10.1145\/375663.375671"},{"issue":"4","key":"9107_CR23","first-page":"33","volume":"23","author":"C. A. Knoblock","year":"2000","unstructured":"Knoblock, C. A., Lerman, K., Minton, S., & Muslea, I. (2000). Accurately and reliably extracting data from the web: A machine learning approach. IEEE Data Engineering Bulletin, 23(4), 33\u201341.","journal-title":"IEEE Data Engineering Bulletin"},{"key":"9107_CR24","doi-asserted-by":"crossref","unstructured":"Lang, K. (1995). Newsweeder: Learning to filter netnews. In Twelfth international conference on machine learning, pp. 331\u2013339.","DOI":"10.1016\/B978-1-55860-377-6.50048-7"},{"key":"9107_CR25","doi-asserted-by":"crossref","unstructured":"Liddle, S. W., Embley, D. W., Scott, D. T., & Yau, S. H. (2002). Extracting data behind web forms, advanced conceptual modeling techniques, pp. 402\u2013413.","DOI":"10.1007\/978-3-540-45275-1_35"},{"key":"9107_CR26","doi-asserted-by":"crossref","unstructured":"Liu, K., Yu, C., & Meng, W. (2002). Discovering the representative of a search engine. In Proceedings of CIKM\u201902, pp. 652\u2013654.","DOI":"10.1145\/584792.584909"},{"key":"9107_CR27","doi-asserted-by":"crossref","unstructured":"Lu, J. (2008). Efficient estimation of the size of text deep web data source. In Proceedings of CIKM 2008, pp. 1485\u20131486.","DOI":"10.1145\/1458082.1458346"},{"key":"9107_CR28","doi-asserted-by":"crossref","unstructured":"Lu, J., Wang, Y., Liang, J., Chen, J., & Liu, J. (2008). An approach to deep web crawling by sampling. In Proceedings of  Web Intelligence, pp. 718\u2013724.","DOI":"10.1109\/WIIAT.2008.392"},{"key":"9107_CR29","doi-asserted-by":"crossref","unstructured":"Nelson, M. L., Smith, J. A., & del Campo, I. G. (2006). Efficient, automatic web resource harvesting. In Proceedings of WIDM\u201906, pp. 43\u201350.","DOI":"10.1145\/1183550.1183560"},{"key":"9107_CR30","doi-asserted-by":"crossref","unstructured":"Ntoulas, A., Zerfos, P., & Cho, J. (2005). Downloading textual hidden web content through keyword queries. In Proceedings of JCDL, 2005, pp. 100\u2013109.","DOI":"10.1145\/1065385.1065407"},{"key":"9107_CR31","first-page":"3","volume":"107","author":"K. H. Pollock","year":"1990","unstructured":"Pollock, K. H., Nichols, J. D., Brownie, C., & Hines, J. E. (1990). Statistical inference for capture crecapture experiments. The Wildlife Society. Wildlife Monographs, 107, 3\u201397.","journal-title":"The Wildlife Society. Wildlife Monographs"},{"key":"9107_CR32","unstructured":"Raghavan, S., & Garcia-Molina, H. (2001). Crawling the hidden web. Proceedings of VLDB 2001."},{"key":"9107_CR35","first-page":"228","volume":"18","author":"F. X. Schumacher","year":"1943","unstructured":"Schumacher, F. X., & Eschmeyer, R. W. (1943). The estimation of fish populations in lakes or ponds. Journal. Tennessee Academy of Science, 18, 228\u2013249.","journal-title":"Journal. Tennessee Academy of Science"},{"issue":"3","key":"9107_CR33","doi-asserted-by":"publisher","first-page":"273","DOI":"10.1016\/S0169-023X(04)00107-7","volume":"52","author":"D. Shestakov","year":"2005","unstructured":"Shestakov, D., Bhowmick, S. S. & Lim, E.-P. (2005). DEQUE: Querying the deep web. Journal of Data & Knowledge Engineering, 52(3), 273\u2013311.","journal-title":"Journal of Data & Knowledge Engineering"},{"key":"9107_CR34","doi-asserted-by":"crossref","unstructured":"Shokouhi, M., Zobel, J., & Scholer, F. (2006). SMM Tahaghoghi, capturing collection size for distributed non-cooperative retrieval. In Proceedings of SIGIR\u201906, pp. 316\u2013323.","DOI":"10.1145\/1148170.1148227"},{"key":"9107_CR36","doi-asserted-by":"crossref","unstructured":"Si, L., & Callan, J. (2003). Relevant document distribution estimation method for resource selection. In Proceedings of SIGIR\u201903.","DOI":"10.1145\/860435.860490"},{"key":"9107_CR37","doi-asserted-by":"crossref","unstructured":"Thomas, P., & Hawking, D. (2007). Evaluating sampling methods for uncooperative collections. In Proceedings of SIGIR, 2007.","DOI":"10.1145\/1277741.1277828"},{"key":"9107_CR38","doi-asserted-by":"crossref","unstructured":"Wu, S., Gibb, F., & Crestani, F. (2003). Experiments with document archive size detection. 25th European conference on IR research, pp. 294\u2013304.","DOI":"10.1007\/3-540-36618-0_21"},{"key":"9107_CR39","unstructured":"Wu, P., Wen, J.-R., Liu, H., & Ma, W.-Y. (2006). Query selection techniques for efficient crawling of structured web sources. In Proceedings of ICDE, 2006, pp. 47\u201356."},{"key":"9107_CR40","doi-asserted-by":"crossref","unstructured":"Xu, J., Wu, S., & Li, X. (2007). Estimating collection size with logistic regression. In Proceedings of SIGIR\u201907, pp. 789\u2013790.","DOI":"10.1145\/1277741.1277910"}],"container-title":["Information Retrieval"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s10791-009-9107-y.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/article\/10.1007\/s10791-009-9107-y\/fulltext.html","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"http:\/\/link.springer.com\/content\/pdf\/10.1007\/s10791-009-9107-y","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"},{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s10791-009-9107-y.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2024,1,2]],"date-time":"2024-01-02T14:47:31Z","timestamp":1704206851000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/s10791-009-9107-y"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2009,8,13]]},"references-count":37,"journal-issue":{"issue":"1","published-print":{"date-parts":[[2010,2]]}},"alternative-id":["9107"],"URL":"https:\/\/doi.org\/10.1007\/s10791-009-9107-y","relation":{},"ISSN":["1386-4564","1573-7659"],"issn-type":[{"value":"1386-4564","type":"print"},{"value":"1573-7659","type":"electronic"}],"subject":[],"published":{"date-parts":[[2009,8,13]]},"assertion":[{"value":"9 April 2009","order":1,"name":"received","label":"Received","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"27 July 2009","order":2,"name":"accepted","label":"Accepted","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"13 August 2009","order":3,"name":"first_online","label":"First Online","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"This content has been made available to all.","name":"free","label":"Free to read"}]}}