Elasticsearch自动补全-ngram

create template

PUT _template/star
{
  "index_patterns":"star*",
  "settings":{
    "refresh_interval":"3s",
    "number_of_replicas":1,
    "number_of_shards":5,
    "analysis":{
      "filter":{
        "auto_complete_filter":{
          "type":"edge_ngram",
          "min_gram":1,
          "max_gram":15
        },
        "pinyin_filter" : {
          "type" : "pinyin",
          "keep_first_letter" : true,
          "keep_full_pinyin" : false,
          "keep_joined_full_pinyin": true,
          "keep_none_chinese" : true,
          "keep_original" : true,
          "limit_first_letter_length" : 16,
          "lowercase" : true,
          "trim_whitespace" : true,
          "keep_none_chinese_in_first_letter" : true
        }
      },
      "analyzer":{
        "chinese_pinyin_prefix_analyzer":{
          "type":"custom",
          "char_filter": [
            "html_strip"
          ],
          "tokenizer":"keyword",
          "filter":[
            "lowercase",
            "pinyin_filter",
            "auto_complete_filter"
          ]
        },
        "chinese_prefix_analyzer":{
          "type":"custom",
          "char_filter": [
            "html_strip"
          ],
          "tokenizer":"keyword",
          "filter":[
            "lowercase",
            "auto_complete_filter"
          ]
        }
      }
    }
  }
}

create index

DELETE star_v1
PUT star_v1/
{
  "mappings": {
    "doc": {
      "properties": {
        "name": {
          "type":  "text",
          "analyzer": "chinese_prefix_analyzer",
          "fields":{
            "prefix":{
              "type":"text",
              "analyzer":"chinese_pinyin_prefix_analyzer"
            }
          }
        }
      }
    }
  }
}

init data

POST star_v1/_bulk/?refresh=true
{ "index" : {"_type" : "doc" } }
{ "name": "刘德华"}
{ "index" : { "_type" : "doc" } }
{ "name": "张学友"}
{ "index" : { "_type" : "doc" } }
{ "name": "gutianle"}
{ "index" : { "_type" : "doc" } }
{ "name": "周杰伦"}
{ "index" : {  "_type" : "doc" } }
{ "name": "林嘉欣"}
{ "index" : {"_type" : "doc" } }
{  "name": "zhangguorong"}
{ "index" : {"_type" : "doc" } }
{  "name": "张杰"}
{ "index" : {"_type" : "doc" } }
{  "name": "任达华"}
{ "index" : {"_type" : "doc" } }
{  "name": "赵微"}
{ "index" : {"_type" : "doc" } }
{  "name": "杨幂"}

suggest for “l”

  • dsl
GET /star_v1/_search
{
  "query": {
    "term": {
      "name.prefix": {
        "value": "l"
      }
    }
  }
}
  • result
{
  "took" : 0,
  "timed_out" : false,
  "_shards" : {
    "total" : 5,
    "successful" : 5,
    "skipped" : 0,
    "failed" : 0
  },
  "hits" : {
    "total" : 3,
    "max_score" : 1.9148799,
    "hits" : [
      {
        "_index" : "star_v1",
        "_type" : "doc",
        "_id" : "_9aqaWsBfEquezXR6q-6",
        "_score" : 1.9148799,
        "_source" : {
          "name" : "刘德华"
        }
      },
      {
        "_index" : "star_v1",
        "_type" : "doc",
        "_id" : "A9aqaWsBfEquezXR6rC6",
        "_score" : 1.9148799,
        "_source" : {
          "name" : "林嘉欣"
        }
      },
      {
        "_index" : "star_v1",
        "_type" : "doc",
        "_id" : "AdaqaWsBfEquezXR6rC6",
        "_score" : 1.038246,
        "_source" : {
          "name" : "gutianle"
        }
      }
    ]
  }
}

suggest for “ld”

  • dsl
GET /star_v1/_search
{
  "query": {
    "term": {
      "name.prefix": {
        "value": "ld"
      }
    }
  }
}
  • resut
{
  "took" : 0,
  "timed_out" : false,
  "_shards" : {
    "total" : 5,
    "successful" : 5,
    "skipped" : 0,
    "failed" : 0
  },
  "hits" : {
    "total" : 1,
    "max_score" : 2.481217,
    "hits" : [
      {
        "_index" : "star_v1",
        "_type" : "doc",
        "_id" : "_9aqaWsBfEquezXR6q-6",
        "_score" : 2.481217,
        "_source" : {
          "name" : "刘德华"
        }
      }
    ]
  }
}

suggest for “刘”

  • dsl
GET /star_v1/_search
{
  "query": {
    "term": {
      "name.prefix": {
        "value": "刘"
      }
    }
  }
}
  • result
{
  "took" : 0,
  "timed_out" : false,
  "_shards" : {
    "total" : 5,
    "successful" : 5,
    "skipped" : 0,
    "failed" : 0
  },
  "hits" : {
    "total" : 1,
    "max_score" : 2.481217,
    "hits" : [
      {
        "_index" : "star_v1",
        "_type" : "doc",
        "_id" : "_9aqaWsBfEquezXR6q-6",
        "_score" : 2.481217,
        "_source" : {
          "name" : "刘德华"
        }
      }
    ]
  }
}

其他

# dsl
POST /star_v1/_analyze
{
  "text": "刘德华",
  "analyzer": "chinese_pinyin_prefix_analyzer"
}

#result
{
  "tokens" : [
    {
      "token" : "刘",
      "start_offset" : 0,
      "end_offset" : 3,
      "type" : "word",
      "position" : 0
    },
    {
      "token" : "刘德",
      "start_offset" : 0,
      "end_offset" : 3,
      "type" : "word",
      "position" : 0
    },
    {
      "token" : "刘德华",
      "start_offset" : 0,
      "end_offset" : 3,
      "type" : "word",
      "position" : 0
    },
    {
      "token" : "l",
      "start_offset" : 0,
      "end_offset" : 3,
      "type" : "word",
      "position" : 0
    },
    {
      "token" : "li",
      "start_offset" : 0,
      "end_offset" : 3,
      "type" : "word",
      "position" : 0
    },
    {
      "token" : "liu",
      "start_offset" : 0,
      "end_offset" : 3,
      "type" : "word",
      "position" : 0
    },
    {
      "token" : "liud",
      "start_offset" : 0,
      "end_offset" : 3,
      "type" : "word",
      "position" : 0
    },
    {
      "token" : "liude",
      "start_offset" : 0,
      "end_offset" : 3,
      "type" : "word",
      "position" : 0
    },
    {
      "token" : "liudeh",
      "start_offset" : 0,
      "end_offset" : 3,
      "type" : "word",
      "position" : 0
    },
    {
      "token" : "liudehu",
      "start_offset" : 0,
      "end_offset" : 3,
      "type" : "word",
      "position" : 0
    },
    {
      "token" : "liudehua",
      "start_offset" : 0,
      "end_offset" : 3,
      "type" : "word",
      "position" : 0
    },
    {
      "token" : "l",
      "start_offset" : 0,
      "end_offset" : 3,
      "type" : "word",
      "position" : 0
    },
    {
      "token" : "ld",
      "start_offset" : 0,
      "end_offset" : 3,
      "type" : "word",
      "position" : 0
    },
    {
      "token" : "ldh",
      "start_offset" : 0,
      "end_offset" : 3,
      "type" : "word",
      "position" : 0
    }
  ]
}

参考

  • 1
    点赞
  • 2
    收藏
    觉得还不错? 一键收藏
  • 0
    评论

“相关推荐”对你有帮助么?

  • 非常没帮助
  • 没帮助
  • 一般
  • 有帮助
  • 非常有帮助
提交
评论
添加红包

请填写红包祝福语或标题

红包个数最小为10个

红包金额最低5元

当前余额3.43前往充值 >
需支付:10.00
成就一亿技术人!
领取后你会自动成为博主和红包主的粉丝 规则
hope_wisdom
发出的红包
实付
使用余额支付
点击重新获取
扫码支付
钱包余额 0

抵扣说明:

1.余额是钱包充值的虚拟货币,按照1:1的比例进行支付金额的抵扣。
2.余额无法直接购买下载,可以购买VIP、付费专栏及课程。

余额充值