2017-12-09 20:28:33 +01:00
|
|
|
// Copyright (c) 2017 Couchbase, Inc.
|
|
|
|
//
|
|
|
|
// Licensed under the Apache License, Version 2.0 (the "License");
|
|
|
|
// you may not use this file except in compliance with the License.
|
|
|
|
// You may obtain a copy of the License at
|
|
|
|
//
|
|
|
|
// http://www.apache.org/licenses/LICENSE-2.0
|
|
|
|
//
|
|
|
|
// Unless required by applicable law or agreed to in writing, software
|
|
|
|
// distributed under the License is distributed on an "AS IS" BASIS,
|
|
|
|
// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
|
|
|
|
// See the License for the specific language governing permissions and
|
|
|
|
// limitations under the License.
|
|
|
|
|
|
|
|
package zap
|
|
|
|
|
|
|
|
import (
|
|
|
|
"encoding/binary"
|
|
|
|
"fmt"
|
|
|
|
|
|
|
|
"github.com/RoaringBitmap/roaring"
|
|
|
|
"github.com/blevesearch/bleve/index"
|
|
|
|
"github.com/blevesearch/bleve/index/scorch/segment"
|
2017-12-19 19:49:57 +01:00
|
|
|
"github.com/couchbase/vellum"
|
|
|
|
"github.com/couchbase/vellum/regexp"
|
2017-12-09 20:28:33 +01:00
|
|
|
)
|
|
|
|
|
|
|
|
// Dictionary is the zap representation of the term dictionary
|
|
|
|
type Dictionary struct {
|
2018-01-18 03:46:57 +01:00
|
|
|
sb *SegmentBase
|
2017-12-09 20:28:33 +01:00
|
|
|
field string
|
|
|
|
fieldID uint16
|
|
|
|
fst *vellum.FST
|
|
|
|
}
|
|
|
|
|
|
|
|
// PostingsList returns the postings list for the specified term
|
|
|
|
func (d *Dictionary) PostingsList(term string, except *roaring.Bitmap) (segment.PostingsList, error) {
|
2018-01-27 18:42:56 +01:00
|
|
|
return d.postingsList([]byte(term), except)
|
2017-12-09 20:28:33 +01:00
|
|
|
}
|
|
|
|
|
2018-01-27 18:42:56 +01:00
|
|
|
func (d *Dictionary) postingsList(term []byte, except *roaring.Bitmap) (*PostingsList, error) {
|
2017-12-09 20:28:33 +01:00
|
|
|
rv := &PostingsList{
|
2018-01-18 03:46:57 +01:00
|
|
|
sb: d.sb,
|
|
|
|
term: term,
|
|
|
|
except: except,
|
2017-12-09 20:28:33 +01:00
|
|
|
}
|
|
|
|
|
|
|
|
if d.fst != nil {
|
2018-01-27 18:42:56 +01:00
|
|
|
postingsOffset, exists, err := d.fst.Get(term)
|
2017-12-09 20:28:33 +01:00
|
|
|
if err != nil {
|
|
|
|
return nil, fmt.Errorf("vellum err: %v", err)
|
|
|
|
}
|
|
|
|
if exists {
|
|
|
|
rv.postingsOffset = postingsOffset
|
|
|
|
// read the location of the freq/norm details
|
|
|
|
var n uint64
|
|
|
|
var read int
|
|
|
|
|
2018-01-18 03:46:57 +01:00
|
|
|
rv.freqOffset, read = binary.Uvarint(d.sb.mem[postingsOffset+n : postingsOffset+binary.MaxVarintLen64])
|
2017-12-09 20:28:33 +01:00
|
|
|
n += uint64(read)
|
2018-01-18 03:46:57 +01:00
|
|
|
rv.locOffset, read = binary.Uvarint(d.sb.mem[postingsOffset+n : postingsOffset+n+binary.MaxVarintLen64])
|
2017-12-09 20:28:33 +01:00
|
|
|
n += uint64(read)
|
2017-12-11 21:59:36 +01:00
|
|
|
|
|
|
|
var locBitmapOffset uint64
|
2018-01-18 03:46:57 +01:00
|
|
|
locBitmapOffset, read = binary.Uvarint(d.sb.mem[postingsOffset+n : postingsOffset+n+binary.MaxVarintLen64])
|
2017-12-11 21:47:41 +01:00
|
|
|
n += uint64(read)
|
2017-12-11 21:59:36 +01:00
|
|
|
|
|
|
|
// go ahead and load loc bitmap
|
|
|
|
var locBitmapLen uint64
|
2018-01-18 03:46:57 +01:00
|
|
|
locBitmapLen, read = binary.Uvarint(d.sb.mem[locBitmapOffset : locBitmapOffset+binary.MaxVarintLen64])
|
|
|
|
locRoaringBytes := d.sb.mem[locBitmapOffset+uint64(read) : locBitmapOffset+uint64(read)+locBitmapLen]
|
2017-12-11 21:59:36 +01:00
|
|
|
rv.locBitmap = roaring.NewBitmap()
|
|
|
|
_, err := rv.locBitmap.FromBuffer(locRoaringBytes)
|
|
|
|
if err != nil {
|
|
|
|
return nil, fmt.Errorf("error loading roaring bitmap of locations with hits: %v", err)
|
|
|
|
}
|
|
|
|
|
2017-12-09 20:28:33 +01:00
|
|
|
var postingsLen uint64
|
2018-01-18 03:46:57 +01:00
|
|
|
postingsLen, read = binary.Uvarint(d.sb.mem[postingsOffset+n : postingsOffset+n+binary.MaxVarintLen64])
|
2017-12-09 20:28:33 +01:00
|
|
|
n += uint64(read)
|
|
|
|
|
2018-01-18 03:46:57 +01:00
|
|
|
roaringBytes := d.sb.mem[postingsOffset+n : postingsOffset+n+postingsLen]
|
2017-12-09 20:28:33 +01:00
|
|
|
|
|
|
|
bitmap := roaring.NewBitmap()
|
|
|
|
_, err = bitmap.FromBuffer(roaringBytes)
|
|
|
|
if err != nil {
|
|
|
|
return nil, fmt.Errorf("error loading roaring bitmap: %v", err)
|
|
|
|
}
|
|
|
|
|
|
|
|
rv.postings = bitmap
|
|
|
|
}
|
|
|
|
}
|
|
|
|
|
|
|
|
return rv, nil
|
|
|
|
}
|
|
|
|
|
|
|
|
// Iterator returns an iterator for this dictionary
|
|
|
|
func (d *Dictionary) Iterator() segment.DictionaryIterator {
|
|
|
|
rv := &DictionaryIterator{
|
|
|
|
d: d,
|
|
|
|
}
|
|
|
|
|
|
|
|
if d.fst != nil {
|
|
|
|
itr, err := d.fst.Iterator(nil, nil)
|
|
|
|
if err == nil {
|
|
|
|
rv.itr = itr
|
|
|
|
}
|
|
|
|
}
|
|
|
|
|
|
|
|
return rv
|
|
|
|
}
|
|
|
|
|
|
|
|
// PrefixIterator returns an iterator which only visits terms having the
|
|
|
|
// the specified prefix
|
|
|
|
func (d *Dictionary) PrefixIterator(prefix string) segment.DictionaryIterator {
|
|
|
|
rv := &DictionaryIterator{
|
|
|
|
d: d,
|
|
|
|
}
|
|
|
|
|
|
|
|
if d.fst != nil {
|
|
|
|
r, err := regexp.New(prefix + ".*")
|
|
|
|
if err == nil {
|
|
|
|
itr, err := d.fst.Search(r, nil, nil)
|
|
|
|
if err == nil {
|
|
|
|
rv.itr = itr
|
|
|
|
}
|
|
|
|
}
|
|
|
|
}
|
|
|
|
|
|
|
|
return rv
|
|
|
|
}
|
|
|
|
|
|
|
|
// RangeIterator returns an iterator which only visits terms between the
|
|
|
|
// start and end terms. NOTE: bleve.index API specifies the end is inclusive.
|
|
|
|
func (d *Dictionary) RangeIterator(start, end string) segment.DictionaryIterator {
|
|
|
|
rv := &DictionaryIterator{
|
|
|
|
d: d,
|
|
|
|
}
|
|
|
|
|
|
|
|
// need to increment the end position to be inclusive
|
|
|
|
endBytes := []byte(end)
|
|
|
|
if endBytes[len(endBytes)-1] < 0xff {
|
|
|
|
endBytes[len(endBytes)-1]++
|
|
|
|
} else {
|
|
|
|
endBytes = append(endBytes, 0xff)
|
|
|
|
}
|
|
|
|
|
|
|
|
if d.fst != nil {
|
|
|
|
itr, err := d.fst.Iterator([]byte(start), endBytes)
|
|
|
|
if err == nil {
|
|
|
|
rv.itr = itr
|
|
|
|
}
|
|
|
|
}
|
|
|
|
|
|
|
|
return rv
|
|
|
|
}
|
|
|
|
|
|
|
|
// DictionaryIterator is an iterator for term dictionary
|
|
|
|
type DictionaryIterator struct {
|
|
|
|
d *Dictionary
|
|
|
|
itr vellum.Iterator
|
|
|
|
err error
|
|
|
|
}
|
|
|
|
|
|
|
|
// Next returns the next entry in the dictionary
|
|
|
|
func (i *DictionaryIterator) Next() (*index.DictEntry, error) {
|
|
|
|
if i.itr == nil || i.err == vellum.ErrIteratorDone {
|
|
|
|
return nil, nil
|
|
|
|
} else if i.err != nil {
|
|
|
|
return nil, i.err
|
|
|
|
}
|
|
|
|
term, count := i.itr.Current()
|
|
|
|
rv := &index.DictEntry{
|
|
|
|
Term: string(term),
|
|
|
|
Count: count,
|
|
|
|
}
|
|
|
|
i.err = i.itr.Next()
|
|
|
|
return rv, nil
|
|
|
|
}
|