利用whoosh对mongoDB的中文文档建立全文检索

1、建立索引

#coding=utf-8

from __future__ import unicode_literals

__author__ = 'zh'

import sys,os

from whoosh.index import create_in,open_dir

from whoosh.fields import *

from jieba.analyse import ChineseAnalyzer

import pymongo

import json

from pymongo.collection import Collection

from pymongo import database

class CreatIndex:

    def __init__(self):

        self.mongoClient = pymongo.MongoClient('192.168.229.128',27017)

        self.websdb = pymongo.database.Database(self.mongoClient,'webdb')

        self.pagesCollection = Collection(self.websdb,'pages')

    def BuiltIndex(self):

        analyzer = ChineseAnalyzer()

        # 索引模版

        schema = Schema(

            U_id=ID(stored=True),

            # md5=ID(stored=True),

            title=TEXT(stored=True,analyzer=analyzer),

            location=TEXT(stored=True),

            publish_time=DATETIME(stored=True,sortable=True),

            content=TEXT(stored=False,analyzer=analyzer)

        )

        from whoosh.filedb.filestore import FileStorage

        storage = FileStorage("../whoosh_index")

        if not os.path.exists("../whoosh_index"):

            os.mkdir("../whoosh_index")

            ix = storage.create_index(schema)

            print '建立索引文件！'

        else:

            ix=storage.open_index()

        # if not os.path.exists("whoosh_index"):

        #     os.mkdir("whoosh_index")

        #     ix = create_in("whoosh_index", schema) # for create new index

        # #ix = open_dir("tmp") # for read only

        writer = ix.writer()

        try:

            num=0

            while(True):

                # break

                try:

                    row=self.pagesCollection.find_one({'indexed':{'$exists':False}})

                    if row!=None:

                        publish_time=None

                        if row.has_key('publish_time'):

                            publish_time=row['publish_time']

                            if str(publish_time)=='' or str(publish_time)=='':

                                publish_time=None

                        location=''

                        if row.has_key('location'):

                            location=json.JSONEncoder().encode(row['location'])

                        writer.add_document(

                        U_id=''.join(str(row['_id'])),

                        # md5=row['md5'],

                        title=row['name'],

                        location=''.join(location),

                        publish_time=publish_time,

                        content=row['information']

                        )

                        self.pagesCollection.update_one({"_id":row["_id"]},{"$set":{"indexed":True}})

                        num+=1

                        print row["_id"],"已建立索引！"

                    else:

                        writer.commit()

                        print "全部处理完毕"

                        # time.sleep(3600)

                        # self.BuiltIndex()

                        break

                except:

                    print row["_id"],"异常"

                    break

        except:

            writer.commit()

            print "异常"

        # print '已处理',num,'共计', self.pagesCollection.find({'indexed':{'$exists':True}}).count()

            print '已处理',num,'共计', self.pagesCollection.find().count()

creatindext = CreatIndex()

creatindext.BuiltIndex()

注：注意编码

2、检索

from __future__ import unicode_literals

#coding=utf-8

__author__ = 'zh'

# from whoosh.qparser import QueryParser

from whoosh import qparser,sorting

# from jieba.analyse import ChineseAnalyzer

from whoosh.index import open_dir

from whoosh.query import *

# import pymongo

import datetime

# from pymongo.collection import Collection

# from pymongo import database

class FullText:

    def __init__(self,index_home='whoosh_index'):

        self.index_home = index_home

        self.ix = open_dir(self.index_home)

        self.searcher = self.ix.searcher()

    # 全文检索,目前主要利用关键字

    def Query(self,parameter):

        # analyzer = ChineseAnalyzer()

        # ix = open_dir(self.index_home) # for read only

        # searcher = ix.searcher()

        # print ix.schema['content']

        # 按照字段查询，可联合查询，MultifieldParser

        list=parameter['keys']

        if len(list)==1:

            parser = qparser.QueryParser(list[0], schema=self.ix.schema)

        if len(list)>1:

            parser = qparser.MultifieldParser(list, schema=self.ix.schema)

        # else:

        #     return None

        # print ix.schema

        keywords = parameter['keywords']

        # print keywords

        q = parser.parse(keywords)

        # mf = sorting.MultiFacet()

        scores = sorting.ScoreFacet()

        date = sorting.FieldFacet("publish_time", reverse=True)

        # 是否分页返回OR全部返回,默认全部返回

        _limit=None

        if parameter.has_key('page') and parameter.has_key('pagesize'):

            page=parameter['page']

            pagesize=parameter['pagesize']

            if page > 0 and pagesize !=0:

                _limit=page*pagesize

        # 是否按照location字段过滤,默认不过滤

        allow_q=None

        if parameter.has_key('includeFields') and parameter['includeFields'].__contains__(u'location'):

            allow_q = qparser.query.Term("location", u"coordinates")

        #  时间分组,暂时不用

        # start = datetime.datetime(2000, 1, 1)

        # end = datetime.datetime.now()

        # gap = datetime.timedelta(days=365)

        # bdayfacet = sorting.DateRangeFacet("publish_time", start, end, gap)

        results = self.searcher.search(q, limit=_limit,filter=allow_q,sortedby=[scores,date])

        # results = searcher.search(q, limit=_limit,filter=restrict_q,

        #                           groupedby=bdayfacet,sortedby=[scores,date])

        # print results.estimated_length()

        return results

fulltext_query = fulltext.FullText()

注：支持多字段检索、分类、排序等

whoosh参考提供陕西省POI数据（300万条，sqlserver备份文件）

利用whoosh对mongoDB的中文文档建立全文检索的更多相关文章

python 搜索引擎Whoosh中文文档和代码以及jieba的使用
注意, 数据库的表最好别有下划线中文文档链接: https://mr-zhao.gitbooks.io/whoosh/content/%E5%A6%82%E4%BD%95%E7%B4%A2%E5%B ...
Phoenix综述（史上最全Phoenix中文文档）
个人主页:http://www.linbingdong.com 简书地址:http://www.jianshu.com/users/6cb45a00b49c/latest_articles 网上关于P ...
Django 1.10中文文档—第一个Django应用Part1
在本教程中,我们将引导您完成一个投票应用程序的创建,它包含下面两部分: 一个可以进行投票和查看结果的公开站点: 一个可以进行增删改查的后台admin管理界面: 我们假设你已经安装了Django.您可以 ...
【Chromium中文文档】OS X 沙箱设计
OS X 沙箱设计转载请注明出处:https://ahangchen.gitbooks.io/chromium_doc_zh/content/zh//General_Architecture/OSX ...
【Chromium中文文档】Chrome/Chromium沙箱 - 安全架构设计
沙箱转载请注明出处:https://ahangchen.gitbooks.io/chromium_doc_zh/content/zh//General_Architecture/Sandbox.ht ...
openstack中文文档
http://www.openstack.cn/p392.html openStack Hacker中文文档 http://docs.mirantis.com/fuel-dev/develop/a ...
【Chromium中文文档】进程模型
进程模型转载请注明出处:https://ahangchen.gitbooks.io/chromium_doc_zh/content/zh//General_Architecture/Process_ ...
【Chromium中文文档】Web安全研究
转载请注明出处:https://ahangchen.gitbooks.io/chromium_doc_zh/content/zh//General_Architecture/Extension_Sec ...
Visual Studio Code中文文档
Visual Studio Code中文文档 Visual Studio Code是一个轻量级但是十分强大的源代码编辑器,重要的是它在Windows, OS X 和Linux操作系统的桌面上均可运行. ...

随机推荐

Java报SQLException
Java报SQLException 具体错误如下: java.sql.SQLException:Column count doesn't match value count at row 1
Java获取某年某周的第一天
Java获取某年某周的第一天 1.设计源码 FirstDayOfWeek.java: /** * @Title:FirstDayOfWeek.java * @Package:com.you.freem ...
Count:2org.apache.batik.transcoder.TranscoderException: null
1.错误描述 Count:2org.apache.batik.transcoder.TranscoderException: null Enclosed Exception: The current ...
学习笔记：webpack深入与实践（一）
一.webpack基本介绍 webpack 是一个现代 JavaScript 应用程序的静态模块打包器(module bundler). 四个核心概念: 入口(entry):指示 webpack 应该 ...
ubuntu安装latex
1 终端中输入"sudo apt-get install texlive-full",输入root密码. 若不想安装所有文件,可以选择"sudo apt-get inst ...
第二篇：使用Spark对MovieLens的特征进行提取
前言在对数据进行了初步探索后,想必读者对MovieLens数据集有了感性认识.而在数据挖掘/推荐引擎运行前,往往需要对数据预处理.预处理的重要性不言而喻,甚至比数据挖掘/推荐系统本身还重要. 然而完 ...
【BZOJ1257】余数之和（数论分块，暴力）
[BZOJ1257]余数之和(数论分块,暴力) 题解 Description 给出正整数n和k,计算j(n, k)=k mod 1 + k mod 2 + k mod 3 + - + k mod n的 ...
[BZOJ2684][CEOI2004]锯木厂选址
BZOJ权限题! Description 从山顶上到山底下沿着一条直线种植了n棵老树.当地的政府决定把他们砍下来.为了不浪费任何一棵木材,树被砍倒后要运送到锯木厂. 木材只能按照一个方向运输:朝山下运 ...
[AH/HNOI2017]单旋
这道题可以用LCT做,开set,LCT,二叉树操作1:直接开set,找到它要插入的位置,一定是前驱,后缀中deep最大的(显然手玩) 操作2:set+LCT询问路径,直接手动提上去,因为树的形态不变 ...
[BZOJ4198] [Noi2015] 荷马史诗 (贪心)
Description 追逐影子的人,自己就是影子. ——荷马 Allison 最近迷上了文学.她喜欢在一个慵懒的午后,细细地品上一杯卡布奇诺,静静地阅读她爱不释手的<荷马史诗>.但是 ...

利用whoosh对mongoDB的中文文档建立全文检索

利用whoosh对mongoDB的中文文档建立全文检索的更多相关文章

随机推荐

热门专题