diff --git a/Godeps/Godeps.json b/Godeps/Godeps.json index 7467e10..ab38ad3 100644 --- a/Godeps/Godeps.json +++ b/Godeps/Godeps.json @@ -10,163 +10,153 @@ }, { "ImportPath": "github.com/blevesearch/bleve", - "Comment": "v0.2.0", - "Rev": "18b50305e4b095808ff138612a5a5cccb7fa40d8" + "Comment": "v0.5.0", + "Rev": "97393d027342f43b17da2f02c090a4e728ac929f" }, { "ImportPath": "github.com/blevesearch/bleve/analysis", - "Comment": "v0.2.0", - "Rev": "18b50305e4b095808ff138612a5a5cccb7fa40d8" + "Comment": "v0.5.0", + "Rev": "97393d027342f43b17da2f02c090a4e728ac929f" }, { "ImportPath": "github.com/blevesearch/bleve/analysis/analyzers/standard_analyzer", - "Comment": "v0.2.0", - "Rev": "18b50305e4b095808ff138612a5a5cccb7fa40d8" - }, - { - "ImportPath": "github.com/blevesearch/bleve/analysis/byte_array_converters/json", - "Comment": "v0.2.0", - "Rev": "18b50305e4b095808ff138612a5a5cccb7fa40d8" + "Comment": "v0.5.0", + "Rev": "97393d027342f43b17da2f02c090a4e728ac929f" }, { "ImportPath": "github.com/blevesearch/bleve/analysis/char_filters/html_char_filter", - "Comment": "v0.2.0", - "Rev": "18b50305e4b095808ff138612a5a5cccb7fa40d8" + "Comment": "v0.5.0", + "Rev": "97393d027342f43b17da2f02c090a4e728ac929f" }, { "ImportPath": "github.com/blevesearch/bleve/analysis/char_filters/regexp_char_filter", - "Comment": "v0.2.0", - "Rev": "18b50305e4b095808ff138612a5a5cccb7fa40d8" + "Comment": "v0.5.0", + "Rev": "97393d027342f43b17da2f02c090a4e728ac929f" }, { "ImportPath": "github.com/blevesearch/bleve/analysis/datetime_parsers/datetime_optional", - "Comment": "v0.2.0", - "Rev": "18b50305e4b095808ff138612a5a5cccb7fa40d8" + "Comment": "v0.5.0", + "Rev": "97393d027342f43b17da2f02c090a4e728ac929f" }, { "ImportPath": "github.com/blevesearch/bleve/analysis/datetime_parsers/flexible_go", - "Comment": "v0.2.0", - "Rev": "18b50305e4b095808ff138612a5a5cccb7fa40d8" + "Comment": "v0.5.0", + "Rev": "97393d027342f43b17da2f02c090a4e728ac929f" }, { "ImportPath": "github.com/blevesearch/bleve/analysis/language/en", - "Comment": "v0.2.0", - "Rev": "18b50305e4b095808ff138612a5a5cccb7fa40d8" + "Comment": "v0.5.0", + "Rev": "97393d027342f43b17da2f02c090a4e728ac929f" }, { "ImportPath": "github.com/blevesearch/bleve/analysis/token_filters/lower_case_filter", - "Comment": "v0.2.0", - "Rev": "18b50305e4b095808ff138612a5a5cccb7fa40d8" + "Comment": "v0.5.0", + "Rev": "97393d027342f43b17da2f02c090a4e728ac929f" }, { "ImportPath": "github.com/blevesearch/bleve/analysis/token_filters/porter", - "Comment": "v0.2.0", - "Rev": "18b50305e4b095808ff138612a5a5cccb7fa40d8" + "Comment": "v0.5.0", + "Rev": "97393d027342f43b17da2f02c090a4e728ac929f" }, { "ImportPath": "github.com/blevesearch/bleve/analysis/token_filters/stop_tokens_filter", - "Comment": "v0.2.0", - "Rev": "18b50305e4b095808ff138612a5a5cccb7fa40d8" + "Comment": "v0.5.0", + "Rev": "97393d027342f43b17da2f02c090a4e728ac929f" }, { "ImportPath": "github.com/blevesearch/bleve/analysis/tokenizers/unicode", - "Comment": "v0.2.0", - "Rev": "18b50305e4b095808ff138612a5a5cccb7fa40d8" + "Comment": "v0.5.0", + "Rev": "97393d027342f43b17da2f02c090a4e728ac929f" }, { "ImportPath": "github.com/blevesearch/bleve/document", - "Comment": "v0.2.0", - "Rev": "18b50305e4b095808ff138612a5a5cccb7fa40d8" + "Comment": "v0.5.0", + "Rev": "97393d027342f43b17da2f02c090a4e728ac929f" }, { "ImportPath": "github.com/blevesearch/bleve/index", - "Comment": "v0.2.0", - "Rev": "18b50305e4b095808ff138612a5a5cccb7fa40d8" - }, - { - "ImportPath": "github.com/blevesearch/bleve/index/firestorm", - "Comment": "v0.2.0", - "Rev": "18b50305e4b095808ff138612a5a5cccb7fa40d8" + "Comment": "v0.5.0", + "Rev": "97393d027342f43b17da2f02c090a4e728ac929f" }, { "ImportPath": "github.com/blevesearch/bleve/index/store", - "Comment": "v0.2.0", - "Rev": "18b50305e4b095808ff138612a5a5cccb7fa40d8" + "Comment": "v0.5.0", + "Rev": "97393d027342f43b17da2f02c090a4e728ac929f" }, { "ImportPath": "github.com/blevesearch/bleve/index/store/boltdb", - "Comment": "v0.2.0", - "Rev": "18b50305e4b095808ff138612a5a5cccb7fa40d8" + "Comment": "v0.5.0", + "Rev": "97393d027342f43b17da2f02c090a4e728ac929f" }, { "ImportPath": "github.com/blevesearch/bleve/index/store/gtreap", - "Comment": "v0.2.0", - "Rev": "18b50305e4b095808ff138612a5a5cccb7fa40d8" + "Comment": "v0.5.0", + "Rev": "97393d027342f43b17da2f02c090a4e728ac929f" }, { "ImportPath": "github.com/blevesearch/bleve/index/upside_down", - "Comment": "v0.2.0", - "Rev": "18b50305e4b095808ff138612a5a5cccb7fa40d8" + "Comment": "v0.5.0", + "Rev": "97393d027342f43b17da2f02c090a4e728ac929f" }, { "ImportPath": "github.com/blevesearch/bleve/numeric_util", - "Comment": "v0.2.0", - "Rev": "18b50305e4b095808ff138612a5a5cccb7fa40d8" + "Comment": "v0.5.0", + "Rev": "97393d027342f43b17da2f02c090a4e728ac929f" }, { "ImportPath": "github.com/blevesearch/bleve/registry", - "Comment": "v0.2.0", - "Rev": "18b50305e4b095808ff138612a5a5cccb7fa40d8" + "Comment": "v0.5.0", + "Rev": "97393d027342f43b17da2f02c090a4e728ac929f" }, { "ImportPath": "github.com/blevesearch/bleve/search", - "Comment": "v0.2.0", - "Rev": "18b50305e4b095808ff138612a5a5cccb7fa40d8" + "Comment": "v0.5.0", + "Rev": "97393d027342f43b17da2f02c090a4e728ac929f" }, { "ImportPath": "github.com/blevesearch/bleve/search/collectors", - "Comment": "v0.2.0", - "Rev": "18b50305e4b095808ff138612a5a5cccb7fa40d8" + "Comment": "v0.5.0", + "Rev": "97393d027342f43b17da2f02c090a4e728ac929f" }, { "ImportPath": "github.com/blevesearch/bleve/search/facets", - "Comment": "v0.2.0", - "Rev": "18b50305e4b095808ff138612a5a5cccb7fa40d8" + "Comment": "v0.5.0", + "Rev": "97393d027342f43b17da2f02c090a4e728ac929f" }, { "ImportPath": "github.com/blevesearch/bleve/search/highlight", - "Comment": "v0.2.0", - "Rev": "18b50305e4b095808ff138612a5a5cccb7fa40d8" + "Comment": "v0.5.0", + "Rev": "97393d027342f43b17da2f02c090a4e728ac929f" }, { "ImportPath": "github.com/blevesearch/bleve/search/highlight/fragment_formatters/html", - "Comment": "v0.2.0", - "Rev": "18b50305e4b095808ff138612a5a5cccb7fa40d8" + "Comment": "v0.5.0", + "Rev": "97393d027342f43b17da2f02c090a4e728ac929f" }, { "ImportPath": "github.com/blevesearch/bleve/search/highlight/fragmenters/simple", - "Comment": "v0.2.0", - "Rev": "18b50305e4b095808ff138612a5a5cccb7fa40d8" + "Comment": "v0.5.0", + "Rev": "97393d027342f43b17da2f02c090a4e728ac929f" }, { "ImportPath": "github.com/blevesearch/bleve/search/highlight/highlighters/html", - "Comment": "v0.2.0", - "Rev": "18b50305e4b095808ff138612a5a5cccb7fa40d8" + "Comment": "v0.5.0", + "Rev": "97393d027342f43b17da2f02c090a4e728ac929f" }, { "ImportPath": "github.com/blevesearch/bleve/search/highlight/highlighters/simple", - "Comment": "v0.2.0", - "Rev": "18b50305e4b095808ff138612a5a5cccb7fa40d8" + "Comment": "v0.5.0", + "Rev": "97393d027342f43b17da2f02c090a4e728ac929f" }, { "ImportPath": "github.com/blevesearch/bleve/search/scorers", - "Comment": "v0.2.0", - "Rev": "18b50305e4b095808ff138612a5a5cccb7fa40d8" + "Comment": "v0.5.0", + "Rev": "97393d027342f43b17da2f02c090a4e728ac929f" }, { "ImportPath": "github.com/blevesearch/bleve/search/searchers", - "Comment": "v0.2.0", - "Rev": "18b50305e4b095808ff138612a5a5cccb7fa40d8" + "Comment": "v0.5.0", + "Rev": "97393d027342f43b17da2f02c090a4e728ac929f" }, { "ImportPath": "github.com/blevesearch/go-porterstemmer", @@ -175,12 +165,12 @@ }, { "ImportPath": "github.com/blevesearch/segment", - "Rev": "db70c57796cc8c310613541dfade3dce627d09c7" + "Rev": "762005e7a34fd909a84586299f1dd457371d36ee" }, { "ImportPath": "github.com/boltdb/bolt", - "Comment": "v1.2.1", - "Rev": "dfb21201d9270c1082d5fb0f07f500311ff72f18" + "Comment": "v1.3.0", + "Rev": "583e8937c61f1af6513608ccc75c97b6abdf4ff9" }, { "ImportPath": "github.com/golang/protobuf/proto", @@ -218,22 +208,17 @@ "ImportPath": "github.com/steveyen/gtreap", "Rev": "0abe01ef9be25c4aedc174758ec2d917314d6d70" }, - { - "ImportPath": "github.com/willf/bitset", - "Comment": "v1.0.0-56-g2e6e809", - "Rev": "2e6e8094ef4745224150c88c16191c7dceaad16f" - }, { "ImportPath": "golang.org/x/crypto/bcrypt", - "Rev": "77f4136a99ffb5ecdbdd0226bd5cb146cf56bc0e" + "Rev": "9477e0b78b9ac3d0b03822fd95422e2fe07627cd" }, { "ImportPath": "golang.org/x/crypto/blowfish", - "Rev": "77f4136a99ffb5ecdbdd0226bd5cb146cf56bc0e" + "Rev": "9477e0b78b9ac3d0b03822fd95422e2fe07627cd" }, { "ImportPath": "golang.org/x/crypto/sha3", - "Rev": "77f4136a99ffb5ecdbdd0226bd5cb146cf56bc0e" + "Rev": "9477e0b78b9ac3d0b03822fd95422e2fe07627cd" }, { "ImportPath": "golang.org/x/net/context", diff --git a/dump.go b/dump.go new file mode 100644 index 0000000..40b92ab --- /dev/null +++ b/dump.go @@ -0,0 +1,76 @@ +package main + +import ( + "encoding/json" + "fmt" + "os" + + "github.com/blevesearch/bleve" +) + +type DumpArticle struct { + Title string `json:"title"` + LinkTitle string `json:"link_title"` + MD string `json:"md"` + Created string `json:"created"` + Updated string `json:"updated"` + Public bool `json:"public"` +} + +func dumpWiki() error { + index, err := bleve.OpenUsing(bleveStore, map[string]interface{}{"read_only": true}) + if err != nil { + return err + } + engine = NewEngine(index) + art, err := engine.Get(false, indexName) + if err != nil { + return err + } + var arts []DumpArticle + titles, err := engine.Dump() + if err != nil { + return err + } + for _, title := range titles { + art, err = engine.Get(false, title) + if err != nil { + return err + } + fmt.Fprintln(os.Stderr, "dump:", art.Title) + arts = append(arts, DumpArticle{ + Title: art.Title, + LinkTitle: art.LinkTitle, + MD: art.MD, + Created: art.Created, + Updated: art.Updated, + Public: art.Public, + }) + } + return json.NewEncoder(os.Stdout).Encode(arts) +} + +func restoreWiki() (err error) { + index, err := bleve.Open(bleveStore) + if err != nil { + mapping := bleve.NewIndexMapping() + index, err = bleve.New(bleveStore, mapping) + if err != nil { + return + } + } + engine = NewEngine(index) + var arts []DumpArticle + err = json.NewDecoder(os.Stdin).Decode(&arts) + if err != nil { + return + } + for _, art := range arts { + fmt.Fprintln(os.Stderr, "restore:", art.Title) + err = engine.Restore(art) + if err != nil { + return + } + } + return +} diff --git a/engine.go b/engine.go index 991dc32..30e585e 100644 --- a/engine.go +++ b/engine.go @@ -4,15 +4,17 @@ import ( "errors" "html/template" "regexp" - "sort" - "strings" "time" "github.com/blevesearch/bleve" "github.com/blevesearch/bleve/document" ) -const timeFmt = "2006-01-02 15:04:05" +const ( + timeFmt = "2006-01-02 15:04:05" + maxAllResult = 1000000 + maxSearchResult = 101 +) var titleRe = regexp.MustCompile("[^a-zA-Z0-9+ ]+") @@ -44,12 +46,21 @@ func (e *Engine) Close() error { return e.index.Close() } +func (e *Engine) Restore(dumpArt DumpArticle) error { + art := render(dumpArt.MD) + art.Title = dumpArt.Title + art.Created = dumpArt.Created + art.Updated = dumpArt.Updated + art.Public = dumpArt.Public + return e.index.Index(dumpArt.LinkTitle, art) +} + func (e *Engine) Index(title, text, user string, public bool) (string, error) { if user == "" { return "", errors.New("engine: user cannot be empty") } title = titleRe.ReplaceAllString(title, "") - newTitle := strings.Replace(title, " ", "+", -1) + newTitle := makeLinkTitle(title) if err := e.Exists(false, newTitle); err == nil { return "", errors.New("engine: article already exists") } @@ -57,8 +68,8 @@ func (e *Engine) Index(title, text, user string, public bool) (string, error) { art.Title = title art.Created = user + " " + now() art.Public = public - e.index.Index(newTitle, art) - return newTitle, nil + err := e.index.Index(newTitle, art) + return newTitle, err } func (e *Engine) Exists(public bool, title string) error { @@ -100,8 +111,7 @@ func (e *Engine) Update(title, text, user string, public bool) error { art.Text = rend.Text art.MD = rend.MD art.Public = public - err = e.index.Index(title, art) - return err + return e.index.Index(title, art) } func (e *Engine) Get(public bool, title string) (Article, error) { @@ -111,7 +121,7 @@ func (e *Engine) Get(public bool, title string) (Article, error) { return art, err } if doc == nil || doc.Fields == nil { - return art, errors.New("engine: article does not esist") + return art, errors.New("engine: article '" + title + "' does not esist") } for _, field := range doc.Fields { switch f := field.(type) { @@ -151,34 +161,33 @@ func (e *Engine) Get(public bool, title string) (Article, error) { func (e *Engine) Search(public bool, search string) (string, error) { var ( res *bleve.SearchResult + req *bleve.SearchRequest err error ) if public { - req := bleve.NewSearchRequest(bleve.NewConjunctionQuery([]bleve.Query{ + req = bleve.NewSearchRequest(bleve.NewConjunctionQuery([]bleve.Query{ bleve.NewBoolFieldQuery(true).SetField("public"), bleve.NewQueryStringQuery("title:" + search + " search:" + search), })) - req.Highlight = bleve.NewHighlightWithStyle("html") - res, err = e.index.Search(req) - if err != nil { - return "", err - } } else { - req := bleve.NewSearchRequest(bleve.NewQueryStringQuery("title:" + search + " search:" + search)) - req.Highlight = bleve.NewHighlightWithStyle("html") - res, err = e.index.Search(req) - if err != nil { - return "", err - } - + req = bleve.NewSearchRequest(bleve.NewQueryStringQuery("title:" + search + " search:" + search)) + } + req.Highlight = bleve.NewHighlightWithStyle("html") + req.Size = maxSearchResult + res, err = e.index.Search(req) + if err != nil { + return "", err } if res.Total == 0 { return "

No search results

", nil - } else if res.Total > 100 { + } else if res.Total == maxSearchResult { return "", errors.New("engine: too many search results") } html := "" for _, hit := range res.Hits { + if hit == nil { + continue + } art, err := e.Get(false, hit.ID) if err != nil { continue @@ -204,29 +213,26 @@ type Title struct { LinkTitle string } -type Titles []Title - -func (t Titles) Len() int { return len(t) } -func (t Titles) Swap(i, j int) { t[i], t[j] = t[j], t[i] } -func (t Titles) Less(i, j int) bool { return t[i].Title < t[j].Title } - -func (e *Engine) GetAll(public bool) (t Titles, err error) { - var res *bleve.SearchResult +func (e *Engine) GetAll(public bool) (t []Title, err error) { + var ( + res *bleve.SearchResult + req *bleve.SearchRequest + ) if public { - req := bleve.NewSearchRequest(bleve.NewBoolFieldQuery(true).SetField("public")) - res, err = e.index.Search(req) - if err != nil { - return - } + req = bleve.NewSearchRequest(bleve.NewBoolFieldQuery(true).SetField("public")) } else { - req := bleve.NewSearchRequest(bleve.NewMatchAllQuery()) - res, err = e.index.Search(req) - if err != nil { - return - } + req = bleve.NewSearchRequest(bleve.NewMatchAllQuery()) + } + req.Size = maxAllResult + res, err = e.index.Search(req) + if err != nil { + return } for _, v := range res.Hits { - if v.ID == "Index" { + if v == nil { + continue + } + if v.ID == indexName { continue } doc, err := e.index.Document(v.ID) @@ -244,6 +250,21 @@ func (e *Engine) GetAll(public bool) (t Titles, err error) { } } } - sort.Sort(t) + return +} + +func (e *Engine) Dump() (titles []string, err error) { + req := bleve.NewSearchRequest(bleve.NewMatchAllQuery()) + req.Size = maxAllResult + res, err := e.index.Search(req) + if err != nil { + return + } + for _, v := range res.Hits { + if v == nil { + continue + } + titles = append(titles, v.ID) + } return } diff --git a/main.go b/main.go index 961471f..6ff7b87 100644 --- a/main.go +++ b/main.go @@ -32,6 +32,7 @@ const ( welcome = `# Welcome to GoWiki This is the [Index](/wiki/Index) page. You can customize it how you like.` + indexName = "Index" ) var ( @@ -44,6 +45,9 @@ var ( engine *Engine db *BoltStore + + dump bool + restore bool ) func main() { @@ -53,8 +57,26 @@ func main() { flag.StringVar(&listen, "l", defListen, "listening
:") flag.StringVar(&logfile, "log", defLog, "log file") flag.BoolVar(&secureCookie, "s", false, "enable secure cookie") + flag.BoolVar(&dump, "dump", false, "create a json dump of all articles in the data directory (dump is written to stdout)") + flag.BoolVar(&restore, "restore", false, "restore a json dump to the data directory (dump is read from stdin)") flag.Parse() + if dump { + err := dumpWiki() + if err != nil { + logger.Fatal("dump: ", err) + } + return + } + + if restore { + err := restoreWiki() + if err != nil { + logger.Fatal("restore: ", err) + } + return + } + cfg := godrop.Config{ User: user, Group: group, @@ -93,7 +115,7 @@ func main() { log.Fatal("main: ", err) } engine = NewEngine(index) - _, err = engine.Index("Index", welcome, defUser, true) + _, err = engine.Index(indexName, welcome, defUser, true) if err != nil { log.Fatal("main: ", err) } diff --git a/render.go b/render.go index 7a63a12..477a813 100644 --- a/render.go +++ b/render.go @@ -20,12 +20,14 @@ func makeFilter() analysis.CharFilter { func render(text string) Article { re := regexp.MustCompile("((.+)()") + link := regexp.MustCompile(`id="(.+)">`) ws := regexp.MustCompile("\\s+") rend := blackfriday.MarkdownCommon([]byte(text)) html := string(rend) index := buildIndex(html) html = strings.Replace(html, "", `
`, -1) html = re.ReplaceAllString(html, `${1} id="${2}">${2}${3}`) + html = link.ReplaceAllStringFunc(html, makeLinkTitle) html = strings.Replace(html, "", `
`, -1) return Article{ Search: ws.ReplaceAllString(string(filter.Filter(rend)), " "), @@ -35,12 +37,17 @@ func render(text string) Article { } } +func makeLinkTitle(title string) string { + return strings.Replace(title, " ", "+", -1) +} + func buildIndex(html string) (ret string) { re := regexp.MustCompile("(.+)") hn := re.FindAllStringSubmatch(html, -1) for k, v := range hn { + link := makeLinkTitle(v[2]) ret += before(hn, k) - ret += `
  • ` + v[2] + "" + ret += `
  • ` + v[2] + "" ret += after(hn, k) } return diff --git a/vendor/github.com/blevesearch/bleve/.gitignore b/vendor/github.com/blevesearch/bleve/.gitignore index b83b503..923b460 100644 --- a/vendor/github.com/blevesearch/bleve/.gitignore +++ b/vendor/github.com/blevesearch/bleve/.gitignore @@ -7,15 +7,10 @@ **/.idea/ **/*.iml .DS_Store +query_string.y.go.tmp /analysis/token_filters/cld2/cld2-read-only /analysis/token_filters/cld2/libcld2_full.a -/utils/bleve_create/bleve_create -/utils/bleve_dump/bleve_dump -/utils/bleve_index/bleve_index -/utils/bleve_bulkindex/bleve_bulkindex -/utils/bleve_index/index.bleve/ -/utils/bleve_query/bleve_query -/utils/bleve_registry/bleve_registry +/cmd/bleve/bleve vendor/** !vendor/manifest /y.output diff --git a/vendor/github.com/blevesearch/bleve/.travis.yml b/vendor/github.com/blevesearch/bleve/.travis.yml index 24d89a2..e24506e 100644 --- a/vendor/github.com/blevesearch/bleve/.travis.yml +++ b/vendor/github.com/blevesearch/bleve/.travis.yml @@ -12,7 +12,6 @@ script: - go get -u github.com/FiloSottile/gvt - gvt restore - go test -v $(go list ./... | grep -v vendor/) - - go test -v ./test -indexType=firestorm - go vet $(go list ./... | grep -v vendor/) - errcheck $(go list ./... | grep -v vendor/) - docs/project-code-coverage.sh diff --git a/vendor/github.com/blevesearch/bleve/CONTRIBUTING.md b/vendor/github.com/blevesearch/bleve/CONTRIBUTING.md new file mode 100644 index 0000000..5ebf3d6 --- /dev/null +++ b/vendor/github.com/blevesearch/bleve/CONTRIBUTING.md @@ -0,0 +1,16 @@ +# Contributing to Bleve + +We look forward to your contributions, but ask that you first review these guidelines. + +### Sign the CLA + +As Bleve is a Couchbase project we require contributors accept the [Couchbase Contributor License Agreement](http://review.couchbase.org/static/individual_agreement.html). To sign this agreement log into the Couchbase [code review tool](http://review.couchbase.org/). The Bleve project does not use this code review tool but it is still used to track acceptance of the contributor license agreements. + +### Submitting a Pull Request + +All types of contributions are welcome, but please keep the following in mind: + +- If you're planning a large change, you should really discuss it in a github issue or on the google group first. This helps avoid duplicate effort and spending time on something that may not be merged. +- Existing tests should continue to pass, new tests for the contribution are nice to have. +- All code should have gone through `go fmt` +- All code should pass `go vet` diff --git a/vendor/github.com/blevesearch/bleve/analysis/byte_array_converters/json/json_byte_array_converter.go b/vendor/github.com/blevesearch/bleve/analysis/byte_array_converters/json/json_byte_array_converter.go deleted file mode 100644 index e07fa4f..0000000 --- a/vendor/github.com/blevesearch/bleve/analysis/byte_array_converters/json/json_byte_array_converter.go +++ /dev/null @@ -1,42 +0,0 @@ -// Copyright (c) 2014 Couchbase, Inc. -// Licensed under the Apache License, Version 2.0 (the "License"); you may not use this file -// except in compliance with the License. You may obtain a copy of the License at -// http://www.apache.org/licenses/LICENSE-2.0 -// Unless required by applicable law or agreed to in writing, software distributed under the -// License is distributed on an "AS IS" BASIS, WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, -// either express or implied. See the License for the specific language governing permissions -// and limitations under the License. - -package json_byte_array_converter - -import ( - "encoding/json" - - "github.com/blevesearch/bleve/analysis" - "github.com/blevesearch/bleve/registry" -) - -const Name = "json" - -type JSONByteArrayConverter struct{} - -func NewJSONByteArrayConverter() *JSONByteArrayConverter { - return &JSONByteArrayConverter{} -} - -func (c *JSONByteArrayConverter) Convert(in []byte) (interface{}, error) { - var rv map[string]interface{} - err := json.Unmarshal(in, &rv) - if err != nil { - return nil, err - } - return rv, nil -} - -func Constructor(config map[string]interface{}, cache *registry.Cache) (analysis.ByteArrayConverter, error) { - return NewJSONByteArrayConverter(), nil -} - -func init() { - registry.RegisterByteArrayConverter(Name, Constructor) -} diff --git a/vendor/github.com/blevesearch/bleve/analysis/language/en/possessive_filter_en.go b/vendor/github.com/blevesearch/bleve/analysis/language/en/possessive_filter_en.go index f322c04..9bae66f 100644 --- a/vendor/github.com/blevesearch/bleve/analysis/language/en/possessive_filter_en.go +++ b/vendor/github.com/blevesearch/bleve/analysis/language/en/possessive_filter_en.go @@ -10,7 +10,7 @@ package en import ( - "bytes" + "unicode/utf8" "github.com/blevesearch/bleve/analysis" "github.com/blevesearch/bleve/registry" @@ -40,15 +40,13 @@ func NewPossessiveFilter() *PossessiveFilter { func (s *PossessiveFilter) Filter(input analysis.TokenStream) analysis.TokenStream { for _, token := range input { - runes := bytes.Runes(token.Term) - if len(runes) >= 2 { - secondToLastRune := runes[len(runes)-2] - lastRune := runes[len(runes)-1] - if (secondToLastRune == rightSingleQuotationMark || - secondToLastRune == apostrophe || - secondToLastRune == fullWidthApostrophe) && - (lastRune == 's' || lastRune == 'S') { - token.Term = analysis.TruncateRunes(token.Term, 2) + lastRune, lastRuneSize := utf8.DecodeLastRune(token.Term) + if lastRune == 's' || lastRune == 'S' { + nextLastRune, nextLastRuneSize := utf8.DecodeLastRune(token.Term[:len(token.Term)-lastRuneSize]) + if nextLastRune == rightSingleQuotationMark || + nextLastRune == apostrophe || + nextLastRune == fullWidthApostrophe { + token.Term = token.Term[:len(token.Term)-lastRuneSize-nextLastRuneSize] } } } diff --git a/vendor/github.com/blevesearch/bleve/analysis/token_filters/porter/porter.go b/vendor/github.com/blevesearch/bleve/analysis/token_filters/porter/porter.go index 7592938..c05bd63 100644 --- a/vendor/github.com/blevesearch/bleve/analysis/token_filters/porter/porter.go +++ b/vendor/github.com/blevesearch/bleve/analysis/token_filters/porter/porter.go @@ -10,6 +10,8 @@ package porter import ( + "bytes" + "github.com/blevesearch/bleve/analysis" "github.com/blevesearch/bleve/registry" @@ -29,8 +31,9 @@ func (s *PorterStemmer) Filter(input analysis.TokenStream) analysis.TokenStream for _, token := range input { // if it is not a protected keyword, stem it if !token.KeyWord { - stemmed := porterstemmer.StemString(string(token.Term)) - token.Term = []byte(stemmed) + termRunes := bytes.Runes(token.Term) + stemmedRunes := porterstemmer.StemWithoutLowerCasing(termRunes) + token.Term = analysis.BuildTermFromRunes(stemmedRunes) } } return input diff --git a/vendor/github.com/blevesearch/bleve/analysis/token_filters/stop_tokens_filter/stop_tokens_filter.go b/vendor/github.com/blevesearch/bleve/analysis/token_filters/stop_tokens_filter/stop_tokens_filter.go index ed9890c..56c3453 100644 --- a/vendor/github.com/blevesearch/bleve/analysis/token_filters/stop_tokens_filter/stop_tokens_filter.go +++ b/vendor/github.com/blevesearch/bleve/analysis/token_filters/stop_tokens_filter/stop_tokens_filter.go @@ -36,16 +36,16 @@ func NewStopTokensFilter(stopTokens analysis.TokenMap) *StopTokensFilter { } func (f *StopTokensFilter) Filter(input analysis.TokenStream) analysis.TokenStream { - rv := make(analysis.TokenStream, 0, len(input)) - + j := 0 for _, token := range input { _, isStopToken := f.stopTokens[string(token.Term)] if !isStopToken { - rv = append(rv, token) + input[j] = token + j++ } } - return rv + return input[:j] } func StopTokensFilterConstructor(config map[string]interface{}, cache *registry.Cache) (analysis.TokenFilter, error) { diff --git a/vendor/github.com/blevesearch/bleve/analysis/util.go b/vendor/github.com/blevesearch/bleve/analysis/util.go index f15f08e..119419c 100644 --- a/vendor/github.com/blevesearch/bleve/analysis/util.go +++ b/vendor/github.com/blevesearch/bleve/analysis/util.go @@ -34,14 +34,32 @@ func InsertRune(in []rune, pos int, r rune) []rune { return rv } -func BuildTermFromRunes(runes []rune) []byte { - rv := make([]byte, 0, len(runes)*4) +// BuildTermFromRunesOptimistic will build a term from the provided runes +// AND optimistically attempt to encode into the provided buffer +// if at any point it appears the buffer is too small, a new buffer is +// allocated and that is used instead +// this should be used in cases where frequently the new term is the same +// length or shorter than the original term (in number of bytes) +func BuildTermFromRunesOptimistic(buf []byte, runes []rune) []byte { + rv := buf + used := 0 for _, r := range runes { - runeBytes := make([]byte, utf8.RuneLen(r)) - utf8.EncodeRune(runeBytes, r) - rv = append(rv, runeBytes...) + nextLen := utf8.RuneLen(r) + if used+nextLen > len(rv) { + // alloc new buf + buf = make([]byte, len(runes)*utf8.UTFMax) + // copy work we've already done + copy(buf, rv[:used]) + rv = buf + } + written := utf8.EncodeRune(rv[used:], r) + used += written } - return rv + return rv[:used] +} + +func BuildTermFromRunes(runes []rune) []byte { + return BuildTermFromRunesOptimistic(make([]byte, len(runes)*utf8.UTFMax), runes) } func TruncateRunes(input []byte, num int) []byte { diff --git a/vendor/github.com/blevesearch/bleve/config.go b/vendor/github.com/blevesearch/bleve/config.go index 135a97a..2a4476c 100644 --- a/vendor/github.com/blevesearch/bleve/config.go +++ b/vendor/github.com/blevesearch/bleve/config.go @@ -15,13 +15,12 @@ import ( "log" "time" + "github.com/blevesearch/bleve/analysis/datetime_parsers/datetime_optional" "github.com/blevesearch/bleve/index" - "github.com/blevesearch/bleve/index/store/boltdb" + "github.com/blevesearch/bleve/index/store/gtreap" "github.com/blevesearch/bleve/index/upside_down" "github.com/blevesearch/bleve/registry" "github.com/blevesearch/bleve/search/highlight/highlighters/html" - - _ "github.com/blevesearch/bleve/index/firestorm" ) var bleveExpVar = expvar.NewMap("bleve") @@ -30,7 +29,9 @@ type configuration struct { Cache *registry.Cache DefaultHighlighter string DefaultKVStore string + DefaultMemKVStore string DefaultIndexType string + QueryDateTimeParser string SlowSearchLogThreshold time.Duration analysisQueue *index.AnalysisQueue } @@ -59,15 +60,23 @@ func init() { Config.DefaultHighlighter = html.Name // default kv store - Config.DefaultKVStore = boltdb.Name + Config.DefaultKVStore = "" + + // default mem only kv store + Config.DefaultMemKVStore = gtreap.Name // default index Config.DefaultIndexType = upside_down.Name + // default query date time parser + Config.QueryDateTimeParser = datetime_optional.Name + bootDuration := time.Since(bootStart) bleveExpVar.Add("bootDuration", int64(bootDuration)) indexStats = NewIndexStats() bleveExpVar.Set("indexes", indexStats) + + initDisk() } var logger = log.New(ioutil.Discard, "bleve", log.LstdFlags) diff --git a/vendor/github.com/blevesearch/bleve/config_app.go b/vendor/github.com/blevesearch/bleve/config_app.go new file mode 100644 index 0000000..ecaa26a --- /dev/null +++ b/vendor/github.com/blevesearch/bleve/config_app.go @@ -0,0 +1,18 @@ +// Copyright (c) 2014 Couchbase, Inc. +// Licensed under the Apache License, Version 2.0 (the "License"); you may not use this file +// except in compliance with the License. You may obtain a copy of the License at +// http://www.apache.org/licenses/LICENSE-2.0 +// Unless required by applicable law or agreed to in writing, software distributed under the +// License is distributed on an "AS IS" BASIS, WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, +// either express or implied. See the License for the specific language governing permissions +// and limitations under the License. + +// +build appengine appenginevm + +package bleve + +// in the appengine environment we cannot support disk based indexes +// so we do no extra configuration in this method +func initDisk() { + +} diff --git a/vendor/github.com/blevesearch/bleve/config_disk.go b/vendor/github.com/blevesearch/bleve/config_disk.go new file mode 100644 index 0000000..8f694a5 --- /dev/null +++ b/vendor/github.com/blevesearch/bleve/config_disk.go @@ -0,0 +1,20 @@ +// Copyright (c) 2014 Couchbase, Inc. +// Licensed under the Apache License, Version 2.0 (the "License"); you may not use this file +// except in compliance with the License. You may obtain a copy of the License at +// http://www.apache.org/licenses/LICENSE-2.0 +// Unless required by applicable law or agreed to in writing, software distributed under the +// License is distributed on an "AS IS" BASIS, WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, +// either express or implied. See the License for the specific language governing permissions +// and limitations under the License. + +// +build !appengine,!appenginevm + +package bleve + +import "github.com/blevesearch/bleve/index/store/boltdb" + +// in normal environments we configure boltdb as the default storage +func initDisk() { + // default kv store + Config.DefaultKVStore = boltdb.Name +} diff --git a/vendor/github.com/blevesearch/bleve/document/document.go b/vendor/github.com/blevesearch/bleve/document/document.go index a596513..db6a1e0 100644 --- a/vendor/github.com/blevesearch/bleve/document/document.go +++ b/vendor/github.com/blevesearch/bleve/document/document.go @@ -9,9 +9,7 @@ package document -import ( - "fmt" -) +import "fmt" type Document struct { ID string `json:"id"` diff --git a/vendor/github.com/blevesearch/bleve/index.go b/vendor/github.com/blevesearch/bleve/index.go index e0b6a3b..df708c6 100644 --- a/vendor/github.com/blevesearch/bleve/index.go +++ b/vendor/github.com/blevesearch/bleve/index.go @@ -176,24 +176,6 @@ type Index interface { FieldDictRange(field string, startTerm []byte, endTerm []byte) (index.FieldDict, error) FieldDictPrefix(field string, termPrefix []byte) (index.FieldDict, error) - // DumpAll returns a channel receiving all index rows as - // UpsideDownCouchRow, in lexicographic byte order. If the enumeration - // fails, an error is sent. The channel is closed once the enumeration - // completes or an error is encountered. The caller must consume all - // channel entries until the channel is closed to ensure the transaction - // and other resources associated with the enumeration are released. - // - // DumpAll exists for debugging and tooling purpose and may change in the - // future. - DumpAll() chan interface{} - - // DumpDoc works like DumpAll but returns only StoredRows and - // TermFrequencyRows related to a document. - DumpDoc(id string) chan interface{} - - // DumpFields works like DumpAll but returns only FieldRows. - DumpFields() chan interface{} - Close() error Mapping() *IndexMapping @@ -228,6 +210,15 @@ func New(path string, mapping *IndexMapping) (Index, error) { return newIndexUsing(path, mapping, Config.DefaultIndexType, Config.DefaultKVStore, nil) } +// NewMemOnly creates a memory-only index. +// The contents of the index is NOT persisted, +// and will be lost once closed. +// The provided mapping will be used for all +// Index/Search operations. +func NewMemOnly(mapping *IndexMapping) (Index, error) { + return newIndexUsing("", mapping, Config.DefaultIndexType, Config.DefaultMemKVStore, nil) +} + // NewUsing creates index at the specified path, // which must not already exist. // The provided mapping will be used for all diff --git a/vendor/github.com/blevesearch/bleve/index/firestorm/analysis.go b/vendor/github.com/blevesearch/bleve/index/firestorm/analysis.go deleted file mode 100644 index 97c093b..0000000 --- a/vendor/github.com/blevesearch/bleve/index/firestorm/analysis.go +++ /dev/null @@ -1,169 +0,0 @@ -// Copyright (c) 2015 Couchbase, Inc. -// Licensed under the Apache License, Version 2.0 (the "License"); you may not use this file -// except in compliance with the License. You may obtain a copy of the License at -// http://www.apache.org/licenses/LICENSE-2.0 -// Unless required by applicable law or agreed to in writing, software distributed under the -// License is distributed on an "AS IS" BASIS, WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, -// either express or implied. See the License for the specific language governing permissions -// and limitations under the License. - -package firestorm - -import ( - "math" - - "github.com/blevesearch/bleve/analysis" - "github.com/blevesearch/bleve/document" - "github.com/blevesearch/bleve/index" -) - -func (f *Firestorm) Analyze(d *document.Document) *index.AnalysisResult { - - rv := &index.AnalysisResult{ - DocID: d.ID, - Rows: make([]index.IndexRow, 0, 100), - } - - docIDBytes := []byte(d.ID) - - // add the _id row - rv.Rows = append(rv.Rows, NewTermFreqRow(0, nil, docIDBytes, d.Number, 0, 0, nil)) - - // information we collate as we merge fields with same name - fieldTermFreqs := make(map[uint16]analysis.TokenFrequencies) - fieldLengths := make(map[uint16]int) - fieldIncludeTermVectors := make(map[uint16]bool) - fieldNames := make(map[uint16]string) - - analyzeField := func(field document.Field, storable bool) { - fieldIndex, newFieldRow := f.fieldIndexOrNewRow(field.Name()) - if newFieldRow != nil { - rv.Rows = append(rv.Rows, newFieldRow) - } - fieldNames[fieldIndex] = field.Name() - - if field.Options().IsIndexed() { - fieldLength, tokenFreqs := field.Analyze() - existingFreqs := fieldTermFreqs[fieldIndex] - if existingFreqs == nil { - fieldTermFreqs[fieldIndex] = tokenFreqs - } else { - existingFreqs.MergeAll(field.Name(), tokenFreqs) - fieldTermFreqs[fieldIndex] = existingFreqs - } - fieldLengths[fieldIndex] += fieldLength - fieldIncludeTermVectors[fieldIndex] = field.Options().IncludeTermVectors() - } - - if storable && field.Options().IsStored() { - storeRow := f.storeField(docIDBytes, d.Number, field, fieldIndex) - rv.Rows = append(rv.Rows, storeRow) - } - } - - for _, field := range d.Fields { - analyzeField(field, true) - } - - if len(d.CompositeFields) > 0 { - for fieldIndex, tokenFreqs := range fieldTermFreqs { - // see if any of the composite fields need this - for _, compositeField := range d.CompositeFields { - compositeField.Compose(fieldNames[fieldIndex], fieldLengths[fieldIndex], tokenFreqs) - } - } - - for _, compositeField := range d.CompositeFields { - analyzeField(compositeField, false) - } - } - - rowsCapNeeded := len(rv.Rows) - for _, tokenFreqs := range fieldTermFreqs { - rowsCapNeeded += len(tokenFreqs) - } - - rows := make([]index.IndexRow, 0, rowsCapNeeded) - rv.Rows = append(rows, rv.Rows...) - - // walk through the collated information and process - // once for each indexed field (unique name) - for fieldIndex, tokenFreqs := range fieldTermFreqs { - fieldLength := fieldLengths[fieldIndex] - includeTermVectors := fieldIncludeTermVectors[fieldIndex] - - rv.Rows = f.indexField(docIDBytes, d.Number, includeTermVectors, fieldIndex, fieldLength, tokenFreqs, rv.Rows) - } - - return rv -} - -func (f *Firestorm) indexField(docID []byte, docNum uint64, includeTermVectors bool, fieldIndex uint16, fieldLength int, tokenFreqs analysis.TokenFrequencies, rows []index.IndexRow) []index.IndexRow { - - tfrs := make([]TermFreqRow, len(tokenFreqs)) - - fieldNorm := float32(1.0 / math.Sqrt(float64(fieldLength))) - - if !includeTermVectors { - i := 0 - for _, tf := range tokenFreqs { - rows = append(rows, InitTermFreqRow(&tfrs[i], fieldIndex, tf.Term, docID, docNum, uint64(tf.Frequency()), fieldNorm, nil)) - i++ - } - return rows - } - - i := 0 - for _, tf := range tokenFreqs { - var tv []*TermVector - tv, rows = f.termVectorsFromTokenFreq(fieldIndex, tf, rows) - rows = append(rows, InitTermFreqRow(&tfrs[i], fieldIndex, tf.Term, docID, docNum, uint64(tf.Frequency()), fieldNorm, tv)) - i++ - } - return rows -} - -func (f *Firestorm) termVectorsFromTokenFreq(field uint16, tf *analysis.TokenFreq, rows []index.IndexRow) ([]*TermVector, []index.IndexRow) { - rv := make([]*TermVector, len(tf.Locations)) - - for i, l := range tf.Locations { - var newFieldRow *FieldRow - fieldIndex := field - if l.Field != "" { - // lookup correct field - fieldIndex, newFieldRow = f.fieldIndexOrNewRow(l.Field) - if newFieldRow != nil { - rows = append(rows, newFieldRow) - } - } - tv := NewTermVector(fieldIndex, uint64(l.Position), uint64(l.Start), uint64(l.End), l.ArrayPositions) - rv[i] = tv - } - - return rv, rows -} - -func (f *Firestorm) storeField(docID []byte, docNum uint64, field document.Field, fieldIndex uint16) index.IndexRow { - fieldValue := make([]byte, 1+len(field.Value())) - fieldValue[0] = encodeFieldType(field) - copy(fieldValue[1:], field.Value()) - storedRow := NewStoredRow(docID, docNum, fieldIndex, field.ArrayPositions(), fieldValue) - return storedRow -} - -func encodeFieldType(f document.Field) byte { - fieldType := byte('x') - switch f.(type) { - case *document.TextField: - fieldType = 't' - case *document.NumericField: - fieldType = 'n' - case *document.DateTimeField: - fieldType = 'd' - case *document.BooleanField: - fieldType = 'b' - case *document.CompositeField: - fieldType = 'c' - } - return fieldType -} diff --git a/vendor/github.com/blevesearch/bleve/index/firestorm/comp.go b/vendor/github.com/blevesearch/bleve/index/firestorm/comp.go deleted file mode 100644 index b411563..0000000 --- a/vendor/github.com/blevesearch/bleve/index/firestorm/comp.go +++ /dev/null @@ -1,156 +0,0 @@ -// Copyright (c) 2015 Couchbase, Inc. -// Licensed under the Apache License, Version 2.0 (the "License"); you may not use this file -// except in compliance with the License. You may obtain a copy of the License at -// http://www.apache.org/licenses/LICENSE-2.0 -// Unless required by applicable law or agreed to in writing, software distributed under the -// License is distributed on an "AS IS" BASIS, WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, -// either express or implied. See the License for the specific language governing permissions -// and limitations under the License. - -package firestorm - -import ( - "bytes" - "math/rand" - "sort" - "sync" - - "github.com/steveyen/gtreap" - "github.com/willf/bitset" -) - -type Compensator struct { - inFlightMutex sync.RWMutex - maxRead uint64 - inFlight *gtreap.Treap - deletedMutex sync.RWMutex - deletedDocNumbers *bitset.BitSet -} - -func NewCompensator() *Compensator { - rv := Compensator{ - inFlight: gtreap.NewTreap(inFlightItemCompare), - deletedDocNumbers: bitset.New(1000000), - } - return &rv -} - -type Snapshot struct { - maxRead uint64 - inFlight *gtreap.Treap - deletedDocNumbers *bitset.BitSet -} - -// returns which doc number is valid -// if none, then 0 -func (s *Snapshot) Which(docID []byte, docNumList DocNumberList) uint64 { - inFlightVal := s.inFlight.Get(&InFlightItem{docID: docID}) - - sort.Sort(docNumList) // Descending ordering. - - for _, docNum := range docNumList { - if docNum > 0 && docNum <= s.maxRead && - (inFlightVal == nil || inFlightVal.(*InFlightItem).docNum == docNum) && - !s.deletedDocNumbers.Test(uint(docNum)) { - return docNum - } - } - return 0 -} - -func (s *Snapshot) Valid(docID []byte, docNum uint64) bool { - logger.Printf("checking validity of: '%s' - % x - %d", docID, docID, docNum) - if docNum > s.maxRead { - return false - } - logger.Printf("<= maxRead") - inFlightVal := s.inFlight.Get(&InFlightItem{docID: docID}) - if inFlightVal != nil && inFlightVal.(*InFlightItem).docNum != docNum { - return false - } - logger.Printf("not in flight") - if s.deletedDocNumbers.Test(uint(docNum)) { - return false - } - logger.Printf("not deleted") - return true -} - -func (c *Compensator) Mutate(docID []byte, docNum uint64) { - c.inFlightMutex.Lock() - defer c.inFlightMutex.Unlock() - c.inFlight = c.inFlight.Upsert(&InFlightItem{docID: docID, docNum: docNum}, rand.Int()) - if docNum != 0 { - c.maxRead = docNum - } -} - -func (c *Compensator) MutateBatch(inflightItems []*InFlightItem, lastDocNum uint64) { - c.inFlightMutex.Lock() - defer c.inFlightMutex.Unlock() - for _, item := range inflightItems { - c.inFlight = c.inFlight.Upsert(item, rand.Int()) - } - c.maxRead = lastDocNum -} - -func (c *Compensator) Migrate(docID []byte, docNum uint64, oldDocNums []uint64) { - c.inFlightMutex.Lock() - defer c.inFlightMutex.Unlock() - c.deletedMutex.Lock() - defer c.deletedMutex.Unlock() - - // clone deleted doc numbers and mutate - if len(oldDocNums) > 0 { - newDeletedDocNumbers := c.deletedDocNumbers.Clone() - for _, oldDocNum := range oldDocNums { - newDeletedDocNumbers.Set(uint(oldDocNum)) - } - // update pointer - c.deletedDocNumbers = newDeletedDocNumbers - } - - // remove entry from in-flight if it still has same doc num - val := c.inFlight.Get(&InFlightItem{docID: docID}) - if val != nil && val.(*InFlightItem).docNum == docNum { - c.inFlight = c.inFlight.Delete(&InFlightItem{docID: docID}) - } -} - -func (c *Compensator) GarbageCollect(docNums []uint64) { - c.deletedMutex.Lock() - defer c.deletedMutex.Unlock() - - for _, docNum := range docNums { - c.deletedDocNumbers.Clear(uint(docNum)) - } -} - -func (c *Compensator) Snapshot() *Snapshot { - c.inFlightMutex.RLock() - defer c.inFlightMutex.RUnlock() - c.deletedMutex.RLock() - defer c.deletedMutex.RUnlock() - - rv := Snapshot{ - maxRead: c.maxRead, - inFlight: c.inFlight, - deletedDocNumbers: c.deletedDocNumbers, - } - return &rv -} - -func (c *Compensator) GarbageCount() uint64 { - return uint64(c.deletedDocNumbers.Count()) -} - -//************** - -type InFlightItem struct { - docID []byte - docNum uint64 -} - -func inFlightItemCompare(a, b interface{}) int { - return bytes.Compare(a.(*InFlightItem).docID, b.(*InFlightItem).docID) -} diff --git a/vendor/github.com/blevesearch/bleve/index/firestorm/dict_updater.go b/vendor/github.com/blevesearch/bleve/index/firestorm/dict_updater.go deleted file mode 100644 index fd598e6..0000000 --- a/vendor/github.com/blevesearch/bleve/index/firestorm/dict_updater.go +++ /dev/null @@ -1,160 +0,0 @@ -// Copyright (c) 2015 Couchbase, Inc. -// Licensed under the Apache License, Version 2.0 (the "License"); you may not use this file -// except in compliance with the License. You may obtain a copy of the License at -// http://www.apache.org/licenses/LICENSE-2.0 -// Unless required by applicable law or agreed to in writing, software distributed under the -// License is distributed on an "AS IS" BASIS, WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, -// either express or implied. See the License for the specific language governing permissions -// and limitations under the License. - -package firestorm - -import ( - "encoding/binary" - "fmt" - "sync" - "sync/atomic" - "time" -) - -const DefaultDictUpdateThreshold = 10 - -var DefaultDictUpdateSleep = 1 * time.Second - -type DictUpdater struct { - batchesStarted uint64 - batchesFlushed uint64 - - f *Firestorm - dictUpdateSleep time.Duration - quit chan struct{} - incoming chan map[string]int64 - - mutex sync.RWMutex - workingSet map[string]int64 - closeWait sync.WaitGroup -} - -func NewDictUpdater(f *Firestorm) *DictUpdater { - rv := DictUpdater{ - f: f, - dictUpdateSleep: DefaultDictUpdateSleep, - workingSet: make(map[string]int64), - batchesStarted: 1, - quit: make(chan struct{}), - incoming: make(chan map[string]int64, 8), - } - return &rv -} - -func (d *DictUpdater) Notify(term string, usage int64) { - d.mutex.Lock() - defer d.mutex.Unlock() - d.workingSet[term] += usage -} - -func (d *DictUpdater) NotifyBatch(termUsages map[string]int64) { - d.incoming <- termUsages -} - -func (d *DictUpdater) Start() { - d.closeWait.Add(1) - go d.runIncoming() - go d.run() -} - -func (d *DictUpdater) Stop() { - close(d.quit) - d.closeWait.Wait() -} - -func (d *DictUpdater) runIncoming() { - for { - select { - case <-d.quit: - return - case termUsages, ok := <-d.incoming: - if !ok { - return - } - d.mutex.Lock() - for term, usage := range termUsages { - d.workingSet[term] += usage - } - d.mutex.Unlock() - } - } -} - -func (d *DictUpdater) run() { - tick := time.Tick(d.dictUpdateSleep) - for { - select { - case <-d.quit: - logger.Printf("dictionary updater asked to quit") - d.closeWait.Done() - return - case <-tick: - logger.Printf("dictionary updater ticked") - d.update() - } - } -} - -func (d *DictUpdater) update() { - d.mutex.Lock() - oldWorkingSet := d.workingSet - d.workingSet = make(map[string]int64) - atomic.AddUint64(&d.batchesStarted, 1) - d.mutex.Unlock() - - // open a writer - writer, err := d.f.store.Writer() - if err != nil { - _ = writer.Close() - logger.Printf("dict updater fatal: %v", err) - return - } - - // prepare batch - wb := writer.NewBatch() - - dictionaryTermDelta := make([]byte, 8) - for term, delta := range oldWorkingSet { - binary.LittleEndian.PutUint64(dictionaryTermDelta, uint64(delta)) - wb.Merge([]byte(term), dictionaryTermDelta) - } - - err = writer.ExecuteBatch(wb) - if err != nil { - _ = writer.Close() - logger.Printf("dict updater fatal: %v", err) - return - } - - atomic.AddUint64(&d.batchesFlushed, 1) - - _ = writer.Close() -} - -// this is not intended to be used publicly, only for unit tests -// which depend on consistency we no longer provide -func (d *DictUpdater) waitTasksDone(dur time.Duration) error { - initial := atomic.LoadUint64(&d.batchesStarted) - timeout := time.After(dur) - tick := time.Tick(100 * time.Millisecond) - for { - select { - // Got a timeout! fail with a timeout error - case <-timeout: - flushed := atomic.LoadUint64(&d.batchesFlushed) - return fmt.Errorf("timeout, %d/%d", initial, flushed) - // Got a tick, we should check on doSomething() - case <-tick: - flushed := atomic.LoadUint64(&d.batchesFlushed) - if flushed > initial { - return nil - } - } - } -} diff --git a/vendor/github.com/blevesearch/bleve/index/firestorm/dictionary.go b/vendor/github.com/blevesearch/bleve/index/firestorm/dictionary.go deleted file mode 100644 index b9f3aba..0000000 --- a/vendor/github.com/blevesearch/bleve/index/firestorm/dictionary.go +++ /dev/null @@ -1,128 +0,0 @@ -// Copyright (c) 2015 Couchbase, Inc. -// Licensed under the Apache License, Version 2.0 (the "License"); you may not use this file -// except in compliance with the License. You may obtain a copy of the License at -// http://www.apache.org/licenses/LICENSE-2.0 -// Unless required by applicable law or agreed to in writing, software distributed under the -// License is distributed on an "AS IS" BASIS, WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, -// either express or implied. See the License for the specific language governing permissions -// and limitations under the License. - -package firestorm - -import ( - "bytes" - "encoding/binary" - "io" - - "github.com/golang/protobuf/proto" -) - -const ByteSeparator byte = 0xff - -var DictionaryKeyPrefix = []byte{'d'} - -type DictionaryRow struct { - field uint16 - term []byte - value DictionaryValue -} - -func NewDictionaryRow(field uint16, term []byte, count uint64) *DictionaryRow { - rv := DictionaryRow{ - field: field, - term: term, - } - rv.value.Count = proto.Uint64(count) - return &rv -} - -func NewDictionaryRowK(key []byte) (*DictionaryRow, error) { - rv := DictionaryRow{} - buf := bytes.NewBuffer(key) - _, err := buf.ReadByte() // type - if err != nil { - return nil, err - } - - err = binary.Read(buf, binary.LittleEndian, &rv.field) - if err != nil { - return nil, err - } - - rv.term, err = buf.ReadBytes(ByteSeparator) - // there is no separator expected here, should get EOF - if err != io.EOF { - return nil, err - } - - return &rv, nil -} - -func (dr *DictionaryRow) parseDictionaryV(value []byte) error { - err := dr.value.Unmarshal(value) - if err != nil { - return err - } - return nil -} - -func NewDictionaryRowKV(key, value []byte) (*DictionaryRow, error) { - rv, err := NewDictionaryRowK(key) - if err != nil { - return nil, err - } - - err = rv.parseDictionaryV(value) - if err != nil { - return nil, err - } - return rv, nil - -} - -func (dr *DictionaryRow) Count() uint64 { - return dr.value.GetCount() -} - -func (dr *DictionaryRow) SetCount(count uint64) { - dr.value.Count = proto.Uint64(count) -} - -func (dr *DictionaryRow) KeySize() int { - return 3 + len(dr.term) -} - -func (dr *DictionaryRow) KeyTo(buf []byte) (int, error) { - copy(buf[0:], DictionaryKeyPrefix) - binary.LittleEndian.PutUint16(buf[1:3], dr.field) - copy(buf[3:], dr.term) - return 3 + len(dr.term), nil -} - -func (dr *DictionaryRow) Key() []byte { - buf := make([]byte, dr.KeySize()) - n, _ := dr.KeyTo(buf) - return buf[:n] -} - -func (dr *DictionaryRow) ValueSize() int { - return dr.value.Size() -} - -func (dr *DictionaryRow) ValueTo(buf []byte) (int, error) { - return dr.value.MarshalTo(buf) -} - -func (dr *DictionaryRow) Value() []byte { - buf := make([]byte, dr.ValueSize()) - n, _ := dr.ValueTo(buf) - return buf[:n] -} - -func DictionaryRowKey(field uint16, term []byte) []byte { - buf := make([]byte, 3+len(term)) - copy(buf[0:], DictionaryKeyPrefix) - binary.LittleEndian.PutUint16(buf[1:3], field) - copy(buf[3:], term) - return buf -} diff --git a/vendor/github.com/blevesearch/bleve/index/firestorm/dump.go b/vendor/github.com/blevesearch/bleve/index/firestorm/dump.go deleted file mode 100644 index b7c222d..0000000 --- a/vendor/github.com/blevesearch/bleve/index/firestorm/dump.go +++ /dev/null @@ -1,92 +0,0 @@ -// Copyright (c) 2015 Couchbase, Inc. -// Licensed under the Apache License, Version 2.0 (the "License"); you may not use this file -// except in compliance with the License. You may obtain a copy of the License at -// http://www.apache.org/licenses/LICENSE-2.0 -// Unless required by applicable law or agreed to in writing, software distributed under the -// License is distributed on an "AS IS" BASIS, WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, -// either express or implied. See the License for the specific language governing permissions -// and limitations under the License. - -package firestorm - -import ( - "bytes" - "fmt" - - "github.com/blevesearch/bleve/index" - "github.com/blevesearch/bleve/index/store" -) - -// the functions in this file are only intended to be used by -// the bleve_dump utility and the debug http handlers -// if your application relies on them, you're doing something wrong -// they may change or be removed at any time - -func (f *Firestorm) dumpPrefix(kvreader store.KVReader, rv chan interface{}, prefix []byte) error { - return visitPrefix(kvreader, prefix, func(key, val []byte) (bool, error) { - row, err := parseFromKeyValue(key, val) - if err != nil { - rv <- err - return false, err - } - rv <- row - return true, nil - }) -} - -func (f *Firestorm) dumpDoc(kvreader store.KVReader, rv chan interface{}, docID []byte) error { - // without a back index we have no choice but to walk the term freq and stored rows - - // walk the term freqs - err := visitPrefix(kvreader, TermFreqKeyPrefix, func(key, val []byte) (bool, error) { - tfr, err := NewTermFreqRowKV(key, val) - if err != nil { - rv <- err - return false, err - } - if bytes.Compare(tfr.DocID(), docID) == 0 { - rv <- tfr - } - return true, nil - }) - - if err != nil { - return err - } - - // now walk the stored - err = visitPrefix(kvreader, StoredKeyPrefix, func(key, val []byte) (bool, error) { - sr, err := NewStoredRowKV(key, val) - if err != nil { - rv <- err - return false, err - } - if bytes.Compare(sr.DocID(), docID) == 0 { - rv <- sr - } - return true, nil - }) - - return err -} - -func parseFromKeyValue(key, value []byte) (index.IndexRow, error) { - if len(key) > 0 { - switch key[0] { - case VersionKey[0]: - return NewVersionRowV(value) - case FieldKeyPrefix[0]: - return NewFieldRowKV(key, value) - case DictionaryKeyPrefix[0]: - return NewDictionaryRowKV(key, value) - case TermFreqKeyPrefix[0]: - return NewTermFreqRowKV(key, value) - case StoredKeyPrefix[0]: - return NewStoredRowKV(key, value) - case InternalKeyPrefix[0]: - return NewInternalRowKV(key, value) - } - return nil, fmt.Errorf("Unknown field type '%s'", string(key[0])) - } - return nil, fmt.Errorf("Invalid empty key") -} diff --git a/vendor/github.com/blevesearch/bleve/index/firestorm/field.go b/vendor/github.com/blevesearch/bleve/index/firestorm/field.go deleted file mode 100644 index e8b7709..0000000 --- a/vendor/github.com/blevesearch/bleve/index/firestorm/field.go +++ /dev/null @@ -1,119 +0,0 @@ -// Copyright (c) 2015 Couchbase, Inc. -// Licensed under the Apache License, Version 2.0 (the "License"); you may not use this file -// except in compliance with the License. You may obtain a copy of the License at -// http://www.apache.org/licenses/LICENSE-2.0 -// Unless required by applicable law or agreed to in writing, software distributed under the -// License is distributed on an "AS IS" BASIS, WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, -// either express or implied. See the License for the specific language governing permissions -// and limitations under the License. - -package firestorm - -import ( - "bytes" - "encoding/binary" - "fmt" - - "github.com/blevesearch/bleve/index/store" - "github.com/golang/protobuf/proto" -) - -var FieldKeyPrefix = []byte{'f'} - -func (f *Firestorm) fieldIndexOrNewRow(name string) (uint16, *FieldRow) { - index, existed := f.fieldCache.FieldNamed(name, true) - if !existed { - return index, NewFieldRow(uint16(index), name) - } - return index, nil -} - -func (f *Firestorm) loadFields(reader store.KVReader) (err error) { - - err = visitPrefix(reader, FieldKeyPrefix, func(key, val []byte) (bool, error) { - fieldRow, err := NewFieldRowKV(key, val) - if err != nil { - return false, err - } - f.fieldCache.AddExisting(fieldRow.Name(), fieldRow.Index()) - return true, nil - }) - - return -} - -type FieldRow struct { - index uint16 - value FieldValue -} - -func NewFieldRow(i uint16, name string) *FieldRow { - rv := FieldRow{ - index: i, - } - rv.value.Name = proto.String(name) - return &rv -} - -func NewFieldRowKV(key, value []byte) (*FieldRow, error) { - rv := FieldRow{} - - buf := bytes.NewBuffer(key) - _, err := buf.ReadByte() // type - if err != nil { - return nil, err - } - err = binary.Read(buf, binary.LittleEndian, &rv.index) - if err != nil { - return nil, err - } - - err = rv.value.Unmarshal(value) - if err != nil { - return nil, err - } - - return &rv, nil -} - -func (fr *FieldRow) KeySize() int { - return 3 -} - -func (fr *FieldRow) KeyTo(buf []byte) (int, error) { - buf[0] = 'f' - binary.LittleEndian.PutUint16(buf[1:3], fr.index) - return 3, nil -} - -func (fr *FieldRow) Key() []byte { - buf := make([]byte, fr.KeySize()) - n, _ := fr.KeyTo(buf) - return buf[:n] -} - -func (fr *FieldRow) ValueSize() int { - return fr.value.Size() -} - -func (fr *FieldRow) ValueTo(buf []byte) (int, error) { - return fr.value.MarshalTo(buf) -} - -func (fr *FieldRow) Value() []byte { - buf := make([]byte, fr.ValueSize()) - n, _ := fr.ValueTo(buf) - return buf[:n] -} - -func (fr *FieldRow) Index() uint16 { - return fr.index -} - -func (fr *FieldRow) Name() string { - return fr.value.GetName() -} - -func (fr *FieldRow) String() string { - return fmt.Sprintf("FieldRow - Field: %d - Name: %s\n", fr.index, fr.Name()) -} diff --git a/vendor/github.com/blevesearch/bleve/index/firestorm/firestorm.go b/vendor/github.com/blevesearch/bleve/index/firestorm/firestorm.go deleted file mode 100644 index abc0784..0000000 --- a/vendor/github.com/blevesearch/bleve/index/firestorm/firestorm.go +++ /dev/null @@ -1,542 +0,0 @@ -// Copyright (c) 2015 Couchbase, Inc. -// Licensed under the Apache License, Version 2.0 (the "License"); you may not use this file -// except in compliance with the License. You may obtain a copy of the License at -// http://www.apache.org/licenses/LICENSE-2.0 -// Unless required by applicable law or agreed to in writing, software distributed under the -// License is distributed on an "AS IS" BASIS, WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, -// either express or implied. See the License for the specific language governing permissions -// and limitations under the License. - -package firestorm - -import ( - "encoding/json" - "sync/atomic" - "time" - - "github.com/blevesearch/bleve/document" - "github.com/blevesearch/bleve/index" - "github.com/blevesearch/bleve/index/store" - "github.com/blevesearch/bleve/registry" -) - -const Name = "firestorm" - -type Firestorm struct { - highDocNumber uint64 - docCount uint64 - - storeName string - storeConfig map[string]interface{} - store store.KVStore - compensator *Compensator - analysisQueue *index.AnalysisQueue - fieldCache *index.FieldCache - garbageCollector *GarbageCollector - lookuper *Lookuper - dictUpdater *DictUpdater - stats *indexStat -} - -func NewFirestorm(storeName string, storeConfig map[string]interface{}, analysisQueue *index.AnalysisQueue) (index.Index, error) { - rv := Firestorm{ - storeName: storeName, - storeConfig: storeConfig, - compensator: NewCompensator(), - analysisQueue: analysisQueue, - fieldCache: index.NewFieldCache(), - docCount: 0, - highDocNumber: 0, - stats: &indexStat{}, - } - rv.stats.f = &rv - rv.garbageCollector = NewGarbageCollector(&rv) - rv.lookuper = NewLookuper(&rv) - rv.dictUpdater = NewDictUpdater(&rv) - return &rv, nil -} - -func (f *Firestorm) Open() (err error) { - - // open the kv store - storeConstructor := registry.KVStoreConstructorByName(f.storeName) - if storeConstructor == nil { - err = index.ErrorUnknownStorageType - return - } - - // now open the store - f.store, err = storeConstructor(&mergeOperator, f.storeConfig) - if err != nil { - return - } - - // start a reader - var kvreader store.KVReader - kvreader, err = f.store.Reader() - if err != nil { - return - } - - // assert correct version, and find out if this is new index - var newIndex bool - newIndex, err = f.checkVersion(kvreader) - if err != nil { - return - } - - if !newIndex { - // process existing index before opening - err = f.warmup(kvreader) - if err != nil { - return - } - } - - err = kvreader.Close() - if err != nil { - return - } - - if newIndex { - // prepare a new index - err = f.bootstrap() - if err != nil { - return - } - } - - // start the garbage collector - f.garbageCollector.Start() - - // start the lookuper - f.lookuper.Start() - - // start the dict updater - f.dictUpdater.Start() - - return -} - -func (f *Firestorm) Close() error { - f.garbageCollector.Stop() - f.lookuper.Stop() - f.dictUpdater.Stop() - return f.store.Close() -} - -func (f *Firestorm) DocCount() (uint64, error) { - count := atomic.LoadUint64(&f.docCount) - return count, nil - -} - -func (f *Firestorm) Update(doc *document.Document) (err error) { - - // assign this document a number - doc.Number = atomic.AddUint64(&f.highDocNumber, 1) - - // do analysis before acquiring write lock - analysisStart := time.Now() - numPlainTextBytes := doc.NumPlainTextBytes() - resultChan := make(chan *index.AnalysisResult) - aw := index.NewAnalysisWork(f, doc, resultChan) - - // put the work on the queue - f.analysisQueue.Queue(aw) - - // wait for the result - result := <-resultChan - close(resultChan) - atomic.AddUint64(&f.stats.analysisTime, uint64(time.Since(analysisStart))) - - // start a writer for this update - indexStart := time.Now() - var kvwriter store.KVWriter - kvwriter, err = f.store.Writer() - if err != nil { - return - } - defer func() { - if cerr := kvwriter.Close(); err == nil && cerr != nil { - err = cerr - } - }() - - var dictionaryDeltas map[string]int64 - dictionaryDeltas, err = f.batchRows(kvwriter, [][]index.IndexRow{result.Rows}, nil) - if err != nil { - _ = kvwriter.Close() - atomic.AddUint64(&f.stats.errors, 1) - return - } - - f.compensator.Mutate([]byte(doc.ID), doc.Number) - f.lookuper.NotifyBatch([]*InFlightItem{{[]byte(doc.ID), doc.Number}}) - f.dictUpdater.NotifyBatch(dictionaryDeltas) - - atomic.AddUint64(&f.stats.indexTime, uint64(time.Since(indexStart))) - atomic.AddUint64(&f.stats.numPlainTextBytesIndexed, numPlainTextBytes) - return -} - -func (f *Firestorm) Delete(id string) error { - indexStart := time.Now() - f.compensator.Mutate([]byte(id), 0) - f.lookuper.NotifyBatch([]*InFlightItem{{[]byte(id), 0}}) - atomic.AddUint64(&f.stats.indexTime, uint64(time.Since(indexStart))) - return nil -} - -func (f *Firestorm) batchRows(writer store.KVWriter, rowsOfRows [][]index.IndexRow, deleteKeys [][]byte) (map[string]int64, error) { - - dictionaryDeltas := make(map[string]int64) - - // count up bytes needed for buffering. - addNum := 0 - addKeyBytes := 0 - addValBytes := 0 - - deleteNum := 0 - deleteKeyBytes := 0 - - var kbuf []byte - - prepareBuf := func(buf []byte, sizeNeeded int) []byte { - if cap(buf) < sizeNeeded { - return make([]byte, sizeNeeded, sizeNeeded+128) - } - return buf[0:sizeNeeded] - } - - for _, rows := range rowsOfRows { - for _, row := range rows { - tfr, ok := row.(*TermFreqRow) - if ok { - if tfr.Field() != 0 { - kbuf = prepareBuf(kbuf, tfr.DictionaryRowKeySize()) - klen, err := tfr.DictionaryRowKeyTo(kbuf) - if err != nil { - return nil, err - } - - dictionaryDeltas[string(kbuf[0:klen])] += 1 - } - } - - addKeyBytes += row.KeySize() - addValBytes += row.ValueSize() - } - addNum += len(rows) - } - - for _, dk := range deleteKeys { - deleteKeyBytes += len(dk) - } - deleteNum += len(deleteKeys) - - // prepare batch - totBytes := addKeyBytes + addValBytes + deleteKeyBytes - - buf, wb, err := writer.NewBatchEx(store.KVBatchOptions{ - TotalBytes: totBytes, - NumSets: addNum, - NumDeletes: deleteNum, - NumMerges: 0, - }) - if err != nil { - return nil, err - } - defer func() { - _ = wb.Close() - }() - - for _, rows := range rowsOfRows { - for _, row := range rows { - klen, err := row.KeyTo(buf) - if err != nil { - return nil, err - } - - vlen, err := row.ValueTo(buf[klen:]) - if err != nil { - return nil, err - } - - wb.Set(buf[0:klen], buf[klen:klen+vlen]) - - buf = buf[klen+vlen:] - } - } - - for _, dk := range deleteKeys { - dklen := copy(buf, dk) - wb.Delete(buf[0:dklen]) - buf = buf[dklen:] - } - - // write out the batch - err = writer.ExecuteBatch(wb) - if err != nil { - return nil, err - } - return dictionaryDeltas, nil -} - -func (f *Firestorm) Batch(batch *index.Batch) (err error) { - - // acquire enough doc numbers for all updates in the batch - // FIXME we actually waste doc numbers because deletes are in the - // same map and we don't need numbers for them - lastDocNumber := atomic.AddUint64(&f.highDocNumber, uint64(len(batch.IndexOps))) - firstDocNumber := lastDocNumber - uint64(len(batch.IndexOps)) + 1 - - analysisStart := time.Now() - resultChan := make(chan *index.AnalysisResult) - - var docsUpdated uint64 - var docsDeleted uint64 - var numPlainTextBytes uint64 - for _, doc := range batch.IndexOps { - if doc != nil { - doc.Number = firstDocNumber // actually assign doc numbers here - firstDocNumber++ - docsUpdated++ - numPlainTextBytes += doc.NumPlainTextBytes() - } else { - docsDeleted++ - } - } - - go func() { - for _, doc := range batch.IndexOps { - if doc != nil { - aw := index.NewAnalysisWork(f, doc, resultChan) - // put the work on the queue - f.analysisQueue.Queue(aw) - } - } - }() - - // extra 1 capacity for internal updates. - collectRows := make([][]index.IndexRow, 0, docsUpdated+1) - - // wait for the result - var itemsDeQueued uint64 - for itemsDeQueued < docsUpdated { - result := <-resultChan - collectRows = append(collectRows, result.Rows) - itemsDeQueued++ - } - close(resultChan) - - atomic.AddUint64(&f.stats.analysisTime, uint64(time.Since(analysisStart))) - - var deleteKeys [][]byte - if len(batch.InternalOps) > 0 { - // add the internal ops - updateInternalRows := make([]index.IndexRow, 0, len(batch.InternalOps)) - for internalKey, internalValue := range batch.InternalOps { - if internalValue == nil { - // delete - deleteInternalRow := NewInternalRow([]byte(internalKey), nil) - deleteKeys = append(deleteKeys, deleteInternalRow.Key()) - } else { - updateInternalRow := NewInternalRow([]byte(internalKey), internalValue) - updateInternalRows = append(updateInternalRows, updateInternalRow) - } - } - collectRows = append(collectRows, updateInternalRows) - } - - inflightItems := make([]*InFlightItem, 0, len(batch.IndexOps)) - for docID, doc := range batch.IndexOps { - if doc != nil { - inflightItems = append(inflightItems, - &InFlightItem{[]byte(docID), doc.Number}) - } else { - inflightItems = append(inflightItems, - &InFlightItem{[]byte(docID), 0}) - } - } - - indexStart := time.Now() - - // start a writer for this batch - var kvwriter store.KVWriter - kvwriter, err = f.store.Writer() - if err != nil { - return - } - - var dictionaryDeltas map[string]int64 - dictionaryDeltas, err = f.batchRows(kvwriter, collectRows, deleteKeys) - if err != nil { - _ = kvwriter.Close() - atomic.AddUint64(&f.stats.errors, 1) - return - } - - f.compensator.MutateBatch(inflightItems, lastDocNumber) - - err = kvwriter.Close() - - f.lookuper.NotifyBatch(inflightItems) - f.dictUpdater.NotifyBatch(dictionaryDeltas) - - atomic.AddUint64(&f.stats.indexTime, uint64(time.Since(indexStart))) - - if err == nil { - atomic.AddUint64(&f.stats.updates, docsUpdated) - atomic.AddUint64(&f.stats.deletes, docsDeleted) - atomic.AddUint64(&f.stats.batches, 1) - atomic.AddUint64(&f.stats.numPlainTextBytesIndexed, numPlainTextBytes) - } else { - atomic.AddUint64(&f.stats.errors, 1) - } - - return -} - -func (f *Firestorm) SetInternal(key, val []byte) (err error) { - internalRow := NewInternalRow(key, val) - var writer store.KVWriter - writer, err = f.store.Writer() - if err != nil { - return - } - defer func() { - if cerr := writer.Close(); err == nil && cerr != nil { - err = cerr - } - }() - - wb := writer.NewBatch() - wb.Set(internalRow.Key(), internalRow.Value()) - - return writer.ExecuteBatch(wb) -} - -func (f *Firestorm) DeleteInternal(key []byte) (err error) { - internalRow := NewInternalRow(key, nil) - var writer store.KVWriter - writer, err = f.store.Writer() - if err != nil { - return - } - defer func() { - if cerr := writer.Close(); err == nil && cerr != nil { - err = cerr - } - }() - - wb := writer.NewBatch() - wb.Delete(internalRow.Key()) - - return writer.ExecuteBatch(wb) -} - -func (f *Firestorm) DumpAll() chan interface{} { - rv := make(chan interface{}) - go func() { - defer close(rv) - - // start an isolated reader for use during the dump - kvreader, err := f.store.Reader() - if err != nil { - rv <- err - return - } - defer func() { - cerr := kvreader.Close() - if cerr != nil { - rv <- cerr - } - }() - - err = f.dumpPrefix(kvreader, rv, nil) - if err != nil { - rv <- err - return - } - }() - return rv -} - -func (f *Firestorm) DumpDoc(docID string) chan interface{} { - rv := make(chan interface{}) - go func() { - defer close(rv) - - // start an isolated reader for use during the dump - kvreader, err := f.store.Reader() - if err != nil { - rv <- err - return - } - defer func() { - cerr := kvreader.Close() - if cerr != nil { - rv <- cerr - } - }() - - err = f.dumpDoc(kvreader, rv, []byte(docID)) - if err != nil { - rv <- err - return - } - }() - return rv -} - -func (f *Firestorm) DumpFields() chan interface{} { - rv := make(chan interface{}) - go func() { - defer close(rv) - - // start an isolated reader for use during the dump - kvreader, err := f.store.Reader() - if err != nil { - rv <- err - return - } - defer func() { - cerr := kvreader.Close() - if cerr != nil { - rv <- cerr - } - }() - - err = f.dumpPrefix(kvreader, rv, FieldKeyPrefix) - if err != nil { - rv <- err - return - } - }() - return rv -} - -func (f *Firestorm) Reader() (index.IndexReader, error) { - return newFirestormReader(f) -} - -func (f *Firestorm) Stats() json.Marshaler { - return f.stats -} - -func (f *Firestorm) StatsMap() map[string]interface{} { - return f.stats.statsMap() -} - -func (f *Firestorm) Wait(timeout time.Duration) error { - return f.dictUpdater.waitTasksDone(timeout) -} - -func (f *Firestorm) Advanced() (store.KVStore, error) { - return f.store, nil -} - -func init() { - registry.RegisterIndexType(Name, NewFirestorm) -} diff --git a/vendor/github.com/blevesearch/bleve/index/firestorm/firestorm.md b/vendor/github.com/blevesearch/bleve/index/firestorm/firestorm.md deleted file mode 100644 index 7e25383..0000000 --- a/vendor/github.com/blevesearch/bleve/index/firestorm/firestorm.md +++ /dev/null @@ -1,382 +0,0 @@ -# Firestorm - -A new indexing scheme for Bleve. - -## Background - -### Goals - -- Avoid a single writer that must pause writing to perform computation - - either by allowing multiple writers, if computation cannot be avoided - - or by having a single writer which can insert rows uninterrupted -- Avoid the need for a back index - - the back index is expensive from a space perspective - - by not writing it out, we should be able to obtain a higher indexing throughput - - consulting the backindex is one of the read/think/update cycles mentioned above - -### Considerations -- The cost for not maintaining a back index is paid in two places - - Searches may need to read more rows, because old/deleted rows may still exist - - These rows can be excluded, so correctness is not affected, but they will be slower - - Old/Deleted rows need to be cleaned up at some point - - This could either be through an explicit cleanup thread, the job of which is to constantly walk the kvstore looking for rows to delete - - Or, it could be integrated with a KV stores natural merge/compaction process (aka RocksDB) - -### Semantics - -It is helpful to review the desired semantics between the Index/Delete operations and Term Searches. - -#### Index(doc_id, doc) - -- Empty Index -- Term Search for "cat" = empty result set - -The Index operation should update the index such that after the operation returns, a matching search would return the document. - -- Index("a", "small cat") -- Term Search for "cat" = {"a"} - -Calling the Index operation again for the same doc_id should update the index such that after the operation returns, only searches matching the newest version return the document. - -- Index("a", "big dog") -- Term Search for "cat" = empty result set -- Term Search for "dog" = {"a"} - -NOTE: - -- At no point during the second index operation would concurrent searches for "cat" and "dog" both return 0 results. -- At no point during the second index operation would concurrent searches for "cat" and "dog" both return 1 result. - -#### Delete(doc_id) - -- Index("a", "small cat") -- Term Search for "cat" = {"a"} -- Delete("a") -- Term Search for "cat" = empty result set - -Once the Delete operation returns, the document should no longer be returned by any search. - -## Details - -### Terminology - -Document ID (`doc_id`) -:The user specified identifier (utf8 string). This never changes for a document. - -Document Number (`doc_number`) -:The Bleve internal identifier (uint64). These numbers are generated from an atomic counter. - -DocIdNumber -: Concatenation of ` 0xff ` - -### Theory of Operation - -By including a new unique identifier as a part of every row generated, the index operation no longer concerns itself with updating existing values or deleting previous values. - -Removal of old rows is handled indepenently by separate threads. - -Ensuring of correct semantics with respect to added/updated/deleted documents is maintained through synchronized in-memory data structures, to compensate for the decoupling of these other operations. - -The Dictionary becomes a best effort data element. In kill-9 scenarios it could become incorrect, but it is believed that this will generally only affect scoring not correctness, and we can pursue read-repair operations. - -### Index State - -The following pseudo-structure will be used to explain changes to the internal state. Keep in mind the datatypes shown represent the logical structure required for correct behavior. The actual implementation may be different to achieve performance goals. - - indexState { - docCount uint64 - fieldCache map[string]uint16 - nextDocNumber uint64 - docIdNumberMutex sync.RWMutex // for protecting fields below - maxReadDocNumber uint64 - inFlightDocIds map[string]uint64 - deletedDocIdNumbers [][]byte - } - -### Operation - -#### Creating New Index - -- New KV Batch -- SET VersionRow{version=X} -- SET FieldRow{field_id=0 field_name="_id"} -- Execute Batch -- Index State intialized to: - - { - docCount = 0 - fieldCache = { - "_id": 0 - } - nextDocNumber = 1 - maxReadDocNumber = 0 - inFlightDocIds = {} - deletedDocIdNumbers = {} - } - -- Garbage Collector Thread is started -- Old Doc Number Lookup Thread is started -- Index marked open - -#### Opening an Existing Index - -- GET VersionRow, assert current version or exit -- ITERATE all FieldRows{} -- ITERATE all TermFrequencyRow{ where field_id = 0 } - - Identify consecutive rows with same doc_id but different doc_number - - Lower document numbers are added to the deletedDocIdNumbers list - - Count all non-duplicate rows, seed the docCount - - Observe highest document number seen, seed nextDocNumber - -- Index State intialized to: - - { - docCount = - fieldCache = { - "_id": 0 - - } - nextDocNumber = + 1 - maxReadDocNumber = - inFlightDocIds = {} - deletedDocIdNumbers = {} - } - -- Garbage Collector Thread is started -- Old Doc Number Lookup Thread is started -- Index marked open - -#### Garbage Collector Thread - -The role of the Garbage Collector thread is to clean up rows referring to document numbers that are no longer relevant (document was deleted or updated). - -Currently, only two types of rows include document numbers: -- Term Frequency Rows -- Stored Rows - -The current thought is that the garbage collector thread will use a single iterator to iterate the following key spaces: - -- TermFrequencyRow { where field_id > 0} -- StoredRow {all} - -For any row refering to a document number on the deletedDocNumbers list, that key will be DELETED. - -The garbage collector will track loop iterations or start key for each deletedDocNumber so that it knows when it has walked a full circle for a given doc number. At point the following happen in order: - -- docNumber is removed from the deletecDocNumbers list -- DELETE is issued on TermFreqRow{ field_id=0, term=doc_id, doc_id=doc_id_number } - -The last thing we do is delete the TermFreqRow for field 0. If anything crashes at any point prior to this, we will again read this record on our next warmup and that doc_id_number will again go through the garbage collection process. - -#### Old Doc Number Lookup Thread - -The role of the Old Doc Number Lookup thread is to asynchronously lookup old document numbers in use for a give document id. - -Waits in a select loop reading from a channel. Through this channel it is notified of a doc_id where work is to be done. When a doc_id comes in, the following is performed: - -- Acquire indexState.docIdNumberMutex for reading: -- Read maxReadDocNumber -- Find doc_id/doc_number k/v pair in the inFlightDocIds map -- Release indexState.docIdNumberMutex -- Start Iterator at TermFrequency{ field_id=0 term=doc_id} -- Iterator until term != doc_id - -All doc_numbers found that are less than maxReadDocNumber and != doc_number in the inFlightDocIds map are now scheduled for deletion. - -- Acquire indexState.docIdNumberMutex for writing: -- add doc numbers to deletedDocIdNumbers -- check if doc_number in inFlightDocIds is still the same - - if so delete it - - if not, it was updated again, so we must leave it -- Release indexState.docIdNumberMutex - -Notify Garbage Collector Thread directly of new doc_numbers. - -#### Term Dictionary Updater Thread - -The role of the Term Dictionary Updater thread is to asynchronously perform best-effort updates to the Term Dictionary. Note the contents of the Term Dictionary only affect scoring, and not correctness of query results. - -NOTE: one case where correctness could be affected is if the dictionary is completely missing a term which has non-zero usage. Since the garbage collector thread is continually looking at these rows, its help could be enlisted to detect/repair this situation. - -It is notified via a channel of increased term usage (by index ops) and of decresed term usage (by garbage collector cleaing up old usage) - -#### Indexing a Document - -- Perform all analysis on the document. -- new_doc_number = indexState.nextDocNumber++ -- Create New Batch -- Batch will contain SET operations for: - - any new Fields - - Term Frequency Rows for indexed fields terms - - Stored Rows for stored fields -- Execute Batch -- Acquire indexState.docIdNumberMutex for writing: -- set maxReadDocNumber new_doc_number -- set inFlightDocIds{ docId = new_doc_number } -- Release indexState.docIdNumberMutex -- Notify Term Frequency Updater thread of increased term usage. -- Notify Old Doc Number Lookup Thread of doc_id. - -The key property is that a search matching the updated document *SHOULD* return the document once this method returns. If the document was an update, it should return the previous document until this method returns. There should be no period of time where neither document matches. - -#### Deleting a Document - -- Acquire indexState.docIdNumberMutex for writing: -- set inFlightDocIds{ docId = 0 } // 0 is a doc number we never use, indicates pending deltion of docId -- Release indexState.docIdNumberMutex -- Notify Old Doc Number Lookup Thread of doc_id. - -#### Batch Operations - -Batch operations look largely just like the indexing/deleting operations. Two other optimizations come into play. - -- More SET operations in the underlying batch -- Larger aggregated updates can be passed to the Term Frequency Updater Thread - -#### Term Field Iteration - -- Acquire indexState.docIdNumberMutex for reading: -- Get copy of: (it is assumed some COW data structure is used, or MVCC is accomodated in some way by the impl) - - maxReadDocNumber - - inFlightDocIds - - deletedDocIdNumbers -- Release indexState.docIdNumberMutex - -Term Field Iteration is used by the basic term search. It produces the set of documents (and related info like term vectors) which used the specified term in the specified field. - -Iterator starts at key: - -```'t' 0xff``` - -Iterator ends when the term does not match. - -- Any row with doc_number > maxReadDocNumber MUST be ignored. -- Any row with doc_id_number on the deletedDocIdNumber list MUST be ignored. -- Any row with the same doc_id as an entry in the inFlightDocIds map, MUST have the same number. - -Any row satisfying the above conditions is a candidate document. - -### Row Encoding - -All keys are manually encoded to ensure a precise row ordering. - -Internal Row values are opaque byte arrays. - -All other values are encoded using protobuf for a balance of efficiency and flexibility. Dictionary and TermFrequency rows are the most likely to take advantage of this flexibility, but other rows are read/written infrequently enough that the flexibility outweighs any overhead. - -#### Version - -There is a single version row which records which version of the firestorm indexing scheme is in use. - -| Key | Value | -|---------|------------| -|```'v'```|``````| - - message VersionValue { - required uint64 version = 1; - } - -#### Field - -Field rows map field names to numeric values - -| Key | Value | -|---------|------------| -|```'f' ```|``````| - - message FieldValue { - required string name = 1; - } - -#### Dictionary - -Dictionary rows record which terms are used in a particular field. The value can be used to store additional information about the term usage. The value will be encoded using protobuf so that future versions can add data to this structure. - -| Key | Value | -|---------|------------| -|```'d' ```|``````| - - message DictionaryValue { - optional uint64 count = 1; // number of documents using this term in this field - } - -#### Term Frequency - -Term Freqquency rows record which documents use a term in a particular field. The value must record how often the term occurs. It may optionally include other details such as a normalization value (precomputed scoring adjustment for the length of the field) and term vectors (where the term occurred within the field). The value will be encoded using protobuf so that future versions can add data to this structure. - -| Key | Value | -|---------|------------| -|```'t' 0xff 0xff ```|``````| - - - message TermVectorEntry { - optional uint32 field = 1; // field optional if redundant, required for composite fields - optional uint64 pos = 2; // positional offset within the field - optional uint64 start = 3; // start byte offset - optional uint64 end = 4; // end byte offset - repeated uint64 arrayPositions = 5; // array positions - } - - message TermFrequencyValue { - required uint64 freq = 1; // frequency of the term occurance within this field - optional float norm = 2; // normalization factor - repeated TermVectorEntry vectors = 3; // term vectors - } - -#### Stored - -Stored rows record the original values used to produce the index. At the row encoding level this is an opaque sequence of bytes. - -| Key | Value | -|---------------------------|-------------------------| -|```'s' 0xff ```|``````| - - message StoredValue { - optional bytes raw = 1; // raw bytes - } - -NOTE: we currently encode stored values as raw bytes, however we have other proposals in flight to do something better than this. By using protobuf here as well, we can support existing functionality through the raw field, but allow for more strongly typed information in the future. - -#### Internal - -Internal rows are a reserved keyspace which the layer above can use for anything it wants. - -| Key | Value | -|---------------------------|-------------------------| -|```'i' ```|``````| - -### FAQ - -1. How do you ensure correct semantics while updating a document in the index? - -Let us consider 5 possible states: - - a. Document X#1 is in the index, maxReadDocNumber=1, inFlightDocIds{}, deletedDocIdNumbers{} - - b. Document X#1 and X#2 are in the index, maxReadDocNumber=1, inFlightDocIds{}, deletedDocIdNumbers{} - - c. Document X#1 and X#2 are in the index, maxReadDocNumber=2, inFlightDocIds{X:2}, deletedDocIdNumbers{} - - d. Document X#1 and X#2 are in the index, maxReadDocNumber=2, inFlightDocIds{}, deletedDocIdNumbers{X#1} - - e. Document X#2 is in the index, maxReadDocNumber=2, inFlightDocIds{}, deletedDocIdNumbers{} - -In state a, we have a steady state where one document has been indexed with id X. - -In state b, we have executed the batch that writes the new rows corresponding to the new version of X, but we have not yet updated our in memory compensation data structures. This is OK, because maxReadDocNumber is still 1, all readers will ignore the new rows we just wrote. This is also OK because we are still inside the Index() method, so there is not yet any expectation to see the udpated document. - -In state c, we have updated both the maxReadDocNumber to 2 and added X:2 to the inFlightDocIds map. This means that searchers could find rows corresponding to X#1 and X#2. However, they are forced to disregard any row for X where the document number is not 2. - -In state d, we have completed the lookup for the old document numbers of X, and found 1. Now deletedDocIdNumbers contains X#1. Now readers that encounter this doc_id_number will ignore it. - -In state e, the garbage collector has removed all record of X#1. - -The Index method returns after it has transitioned to state c, which maintains the semantics we desire. - -2\. Wait, what happens if I kill -9 the process, won't you forget about the deleted documents? - -No, our proposal is for a warmup process to walk a subset of the keyspace (TermFreq{ where field_id=0 }). This warmup process will identify all not-yet cleaned up document numbers, and seed the deletedDocIdNumbers state as well as the Garbage Collector Thread. - -3\. Wait, but what will happen to the inFlightDocIds in a kill -9 scenario? - -It turns out they actually don't matter. That list was just an optimization to get us through the window of time while we hadn't yet looked up the old document numbers for a given document id. But, during the warmup phase we still identify all those keys and they go directly onto deletedDocIdNumbers list. \ No newline at end of file diff --git a/vendor/github.com/blevesearch/bleve/index/firestorm/firestorm_rows.pb.go b/vendor/github.com/blevesearch/bleve/index/firestorm/firestorm_rows.pb.go deleted file mode 100644 index f8add2c..0000000 --- a/vendor/github.com/blevesearch/bleve/index/firestorm/firestorm_rows.pb.go +++ /dev/null @@ -1,1125 +0,0 @@ -// Code generated by protoc-gen-gogo. -// source: firestorm_rows.proto -// DO NOT EDIT! - -/* - Package firestorm is a generated protocol buffer package. - - It is generated from these files: - firestorm_rows.proto - - It has these top-level messages: - VersionValue - FieldValue - DictionaryValue - TermVector - TermFreqValue - StoredValue -*/ -package firestorm - -import proto "github.com/golang/protobuf/proto" -import math "math" - -import io "io" -import fmt "fmt" -import github_com_golang_protobuf_proto "github.com/golang/protobuf/proto" - -// Reference imports to suppress errors if they are not otherwise used. -var _ = proto.Marshal -var _ = math.Inf - -type VersionValue struct { - Version *uint64 `protobuf:"varint,1,req,name=version" json:"version,omitempty"` - XXX_unrecognized []byte `json:"-"` -} - -func (m *VersionValue) Reset() { *m = VersionValue{} } -func (m *VersionValue) String() string { return proto.CompactTextString(m) } -func (*VersionValue) ProtoMessage() {} - -func (m *VersionValue) GetVersion() uint64 { - if m != nil && m.Version != nil { - return *m.Version - } - return 0 -} - -type FieldValue struct { - Name *string `protobuf:"bytes,1,req,name=name" json:"name,omitempty"` - XXX_unrecognized []byte `json:"-"` -} - -func (m *FieldValue) Reset() { *m = FieldValue{} } -func (m *FieldValue) String() string { return proto.CompactTextString(m) } -func (*FieldValue) ProtoMessage() {} - -func (m *FieldValue) GetName() string { - if m != nil && m.Name != nil { - return *m.Name - } - return "" -} - -type DictionaryValue struct { - Count *uint64 `protobuf:"varint,1,opt,name=count" json:"count,omitempty"` - XXX_unrecognized []byte `json:"-"` -} - -func (m *DictionaryValue) Reset() { *m = DictionaryValue{} } -func (m *DictionaryValue) String() string { return proto.CompactTextString(m) } -func (*DictionaryValue) ProtoMessage() {} - -func (m *DictionaryValue) GetCount() uint64 { - if m != nil && m.Count != nil { - return *m.Count - } - return 0 -} - -type TermVector struct { - Field *uint32 `protobuf:"varint,1,opt,name=field" json:"field,omitempty"` - Pos *uint64 `protobuf:"varint,2,opt,name=pos" json:"pos,omitempty"` - Start *uint64 `protobuf:"varint,3,opt,name=start" json:"start,omitempty"` - End *uint64 `protobuf:"varint,4,opt,name=end" json:"end,omitempty"` - ArrayPositions []uint64 `protobuf:"varint,5,rep,name=arrayPositions" json:"arrayPositions,omitempty"` - XXX_unrecognized []byte `json:"-"` -} - -func (m *TermVector) Reset() { *m = TermVector{} } -func (m *TermVector) String() string { return proto.CompactTextString(m) } -func (*TermVector) ProtoMessage() {} - -func (m *TermVector) GetField() uint32 { - if m != nil && m.Field != nil { - return *m.Field - } - return 0 -} - -func (m *TermVector) GetPos() uint64 { - if m != nil && m.Pos != nil { - return *m.Pos - } - return 0 -} - -func (m *TermVector) GetStart() uint64 { - if m != nil && m.Start != nil { - return *m.Start - } - return 0 -} - -func (m *TermVector) GetEnd() uint64 { - if m != nil && m.End != nil { - return *m.End - } - return 0 -} - -func (m *TermVector) GetArrayPositions() []uint64 { - if m != nil { - return m.ArrayPositions - } - return nil -} - -type TermFreqValue struct { - Freq *uint64 `protobuf:"varint,1,req,name=freq" json:"freq,omitempty"` - Norm *float32 `protobuf:"fixed32,2,opt,name=norm" json:"norm,omitempty"` - Vectors []*TermVector `protobuf:"bytes,3,rep,name=vectors" json:"vectors,omitempty"` - XXX_unrecognized []byte `json:"-"` -} - -func (m *TermFreqValue) Reset() { *m = TermFreqValue{} } -func (m *TermFreqValue) String() string { return proto.CompactTextString(m) } -func (*TermFreqValue) ProtoMessage() {} - -func (m *TermFreqValue) GetFreq() uint64 { - if m != nil && m.Freq != nil { - return *m.Freq - } - return 0 -} - -func (m *TermFreqValue) GetNorm() float32 { - if m != nil && m.Norm != nil { - return *m.Norm - } - return 0 -} - -func (m *TermFreqValue) GetVectors() []*TermVector { - if m != nil { - return m.Vectors - } - return nil -} - -type StoredValue struct { - Raw []byte `protobuf:"bytes,1,opt,name=raw" json:"raw,omitempty"` - XXX_unrecognized []byte `json:"-"` -} - -func (m *StoredValue) Reset() { *m = StoredValue{} } -func (m *StoredValue) String() string { return proto.CompactTextString(m) } -func (*StoredValue) ProtoMessage() {} - -func (m *StoredValue) GetRaw() []byte { - if m != nil { - return m.Raw - } - return nil -} - -func (m *VersionValue) Unmarshal(data []byte) error { - var hasFields [1]uint64 - l := len(data) - iNdEx := 0 - for iNdEx < l { - var wire uint64 - for shift := uint(0); ; shift += 7 { - if iNdEx >= l { - return io.ErrUnexpectedEOF - } - b := data[iNdEx] - iNdEx++ - wire |= (uint64(b) & 0x7F) << shift - if b < 0x80 { - break - } - } - fieldNum := int32(wire >> 3) - wireType := int(wire & 0x7) - switch fieldNum { - case 1: - if wireType != 0 { - return fmt.Errorf("proto: wrong wireType = %d for field Version", wireType) - } - var v uint64 - for shift := uint(0); ; shift += 7 { - if iNdEx >= l { - return io.ErrUnexpectedEOF - } - b := data[iNdEx] - iNdEx++ - v |= (uint64(b) & 0x7F) << shift - if b < 0x80 { - break - } - } - m.Version = &v - hasFields[0] |= uint64(0x00000001) - default: - var sizeOfWire int - for { - sizeOfWire++ - wire >>= 7 - if wire == 0 { - break - } - } - iNdEx -= sizeOfWire - skippy, err := skipFirestormRows(data[iNdEx:]) - if err != nil { - return err - } - if skippy < 0 { - return ErrInvalidLengthFirestormRows - } - if (iNdEx + skippy) > l { - return io.ErrUnexpectedEOF - } - m.XXX_unrecognized = append(m.XXX_unrecognized, data[iNdEx:iNdEx+skippy]...) - iNdEx += skippy - } - } - if hasFields[0]&uint64(0x00000001) == 0 { - return new(github_com_golang_protobuf_proto.RequiredNotSetError) - } - - return nil -} -func (m *FieldValue) Unmarshal(data []byte) error { - var hasFields [1]uint64 - l := len(data) - iNdEx := 0 - for iNdEx < l { - var wire uint64 - for shift := uint(0); ; shift += 7 { - if iNdEx >= l { - return io.ErrUnexpectedEOF - } - b := data[iNdEx] - iNdEx++ - wire |= (uint64(b) & 0x7F) << shift - if b < 0x80 { - break - } - } - fieldNum := int32(wire >> 3) - wireType := int(wire & 0x7) - switch fieldNum { - case 1: - if wireType != 2 { - return fmt.Errorf("proto: wrong wireType = %d for field Name", wireType) - } - var stringLen uint64 - for shift := uint(0); ; shift += 7 { - if iNdEx >= l { - return io.ErrUnexpectedEOF - } - b := data[iNdEx] - iNdEx++ - stringLen |= (uint64(b) & 0x7F) << shift - if b < 0x80 { - break - } - } - postIndex := iNdEx + int(stringLen) - if postIndex > l { - return io.ErrUnexpectedEOF - } - s := string(data[iNdEx:postIndex]) - m.Name = &s - iNdEx = postIndex - hasFields[0] |= uint64(0x00000001) - default: - var sizeOfWire int - for { - sizeOfWire++ - wire >>= 7 - if wire == 0 { - break - } - } - iNdEx -= sizeOfWire - skippy, err := skipFirestormRows(data[iNdEx:]) - if err != nil { - return err - } - if skippy < 0 { - return ErrInvalidLengthFirestormRows - } - if (iNdEx + skippy) > l { - return io.ErrUnexpectedEOF - } - m.XXX_unrecognized = append(m.XXX_unrecognized, data[iNdEx:iNdEx+skippy]...) - iNdEx += skippy - } - } - if hasFields[0]&uint64(0x00000001) == 0 { - return new(github_com_golang_protobuf_proto.RequiredNotSetError) - } - - return nil -} -func (m *DictionaryValue) Unmarshal(data []byte) error { - l := len(data) - iNdEx := 0 - for iNdEx < l { - var wire uint64 - for shift := uint(0); ; shift += 7 { - if iNdEx >= l { - return io.ErrUnexpectedEOF - } - b := data[iNdEx] - iNdEx++ - wire |= (uint64(b) & 0x7F) << shift - if b < 0x80 { - break - } - } - fieldNum := int32(wire >> 3) - wireType := int(wire & 0x7) - switch fieldNum { - case 1: - if wireType != 0 { - return fmt.Errorf("proto: wrong wireType = %d for field Count", wireType) - } - var v uint64 - for shift := uint(0); ; shift += 7 { - if iNdEx >= l { - return io.ErrUnexpectedEOF - } - b := data[iNdEx] - iNdEx++ - v |= (uint64(b) & 0x7F) << shift - if b < 0x80 { - break - } - } - m.Count = &v - default: - var sizeOfWire int - for { - sizeOfWire++ - wire >>= 7 - if wire == 0 { - break - } - } - iNdEx -= sizeOfWire - skippy, err := skipFirestormRows(data[iNdEx:]) - if err != nil { - return err - } - if skippy < 0 { - return ErrInvalidLengthFirestormRows - } - if (iNdEx + skippy) > l { - return io.ErrUnexpectedEOF - } - m.XXX_unrecognized = append(m.XXX_unrecognized, data[iNdEx:iNdEx+skippy]...) - iNdEx += skippy - } - } - - return nil -} -func (m *TermVector) Unmarshal(data []byte) error { - l := len(data) - iNdEx := 0 - for iNdEx < l { - var wire uint64 - for shift := uint(0); ; shift += 7 { - if iNdEx >= l { - return io.ErrUnexpectedEOF - } - b := data[iNdEx] - iNdEx++ - wire |= (uint64(b) & 0x7F) << shift - if b < 0x80 { - break - } - } - fieldNum := int32(wire >> 3) - wireType := int(wire & 0x7) - switch fieldNum { - case 1: - if wireType != 0 { - return fmt.Errorf("proto: wrong wireType = %d for field Field", wireType) - } - var v uint32 - for shift := uint(0); ; shift += 7 { - if iNdEx >= l { - return io.ErrUnexpectedEOF - } - b := data[iNdEx] - iNdEx++ - v |= (uint32(b) & 0x7F) << shift - if b < 0x80 { - break - } - } - m.Field = &v - case 2: - if wireType != 0 { - return fmt.Errorf("proto: wrong wireType = %d for field Pos", wireType) - } - var v uint64 - for shift := uint(0); ; shift += 7 { - if iNdEx >= l { - return io.ErrUnexpectedEOF - } - b := data[iNdEx] - iNdEx++ - v |= (uint64(b) & 0x7F) << shift - if b < 0x80 { - break - } - } - m.Pos = &v - case 3: - if wireType != 0 { - return fmt.Errorf("proto: wrong wireType = %d for field Start", wireType) - } - var v uint64 - for shift := uint(0); ; shift += 7 { - if iNdEx >= l { - return io.ErrUnexpectedEOF - } - b := data[iNdEx] - iNdEx++ - v |= (uint64(b) & 0x7F) << shift - if b < 0x80 { - break - } - } - m.Start = &v - case 4: - if wireType != 0 { - return fmt.Errorf("proto: wrong wireType = %d for field End", wireType) - } - var v uint64 - for shift := uint(0); ; shift += 7 { - if iNdEx >= l { - return io.ErrUnexpectedEOF - } - b := data[iNdEx] - iNdEx++ - v |= (uint64(b) & 0x7F) << shift - if b < 0x80 { - break - } - } - m.End = &v - case 5: - if wireType != 0 { - return fmt.Errorf("proto: wrong wireType = %d for field ArrayPositions", wireType) - } - var v uint64 - for shift := uint(0); ; shift += 7 { - if iNdEx >= l { - return io.ErrUnexpectedEOF - } - b := data[iNdEx] - iNdEx++ - v |= (uint64(b) & 0x7F) << shift - if b < 0x80 { - break - } - } - m.ArrayPositions = append(m.ArrayPositions, v) - default: - var sizeOfWire int - for { - sizeOfWire++ - wire >>= 7 - if wire == 0 { - break - } - } - iNdEx -= sizeOfWire - skippy, err := skipFirestormRows(data[iNdEx:]) - if err != nil { - return err - } - if skippy < 0 { - return ErrInvalidLengthFirestormRows - } - if (iNdEx + skippy) > l { - return io.ErrUnexpectedEOF - } - m.XXX_unrecognized = append(m.XXX_unrecognized, data[iNdEx:iNdEx+skippy]...) - iNdEx += skippy - } - } - - return nil -} -func (m *TermFreqValue) Unmarshal(data []byte) error { - var hasFields [1]uint64 - l := len(data) - iNdEx := 0 - for iNdEx < l { - var wire uint64 - for shift := uint(0); ; shift += 7 { - if iNdEx >= l { - return io.ErrUnexpectedEOF - } - b := data[iNdEx] - iNdEx++ - wire |= (uint64(b) & 0x7F) << shift - if b < 0x80 { - break - } - } - fieldNum := int32(wire >> 3) - wireType := int(wire & 0x7) - switch fieldNum { - case 1: - if wireType != 0 { - return fmt.Errorf("proto: wrong wireType = %d for field Freq", wireType) - } - var v uint64 - for shift := uint(0); ; shift += 7 { - if iNdEx >= l { - return io.ErrUnexpectedEOF - } - b := data[iNdEx] - iNdEx++ - v |= (uint64(b) & 0x7F) << shift - if b < 0x80 { - break - } - } - m.Freq = &v - hasFields[0] |= uint64(0x00000001) - case 2: - if wireType != 5 { - return fmt.Errorf("proto: wrong wireType = %d for field Norm", wireType) - } - var v uint32 - if (iNdEx + 4) > l { - return io.ErrUnexpectedEOF - } - iNdEx += 4 - v = uint32(data[iNdEx-4]) - v |= uint32(data[iNdEx-3]) << 8 - v |= uint32(data[iNdEx-2]) << 16 - v |= uint32(data[iNdEx-1]) << 24 - v2 := float32(math.Float32frombits(v)) - m.Norm = &v2 - case 3: - if wireType != 2 { - return fmt.Errorf("proto: wrong wireType = %d for field Vectors", wireType) - } - var msglen int - for shift := uint(0); ; shift += 7 { - if iNdEx >= l { - return io.ErrUnexpectedEOF - } - b := data[iNdEx] - iNdEx++ - msglen |= (int(b) & 0x7F) << shift - if b < 0x80 { - break - } - } - postIndex := iNdEx + msglen - if msglen < 0 { - return ErrInvalidLengthFirestormRows - } - if postIndex > l { - return io.ErrUnexpectedEOF - } - m.Vectors = append(m.Vectors, &TermVector{}) - if err := m.Vectors[len(m.Vectors)-1].Unmarshal(data[iNdEx:postIndex]); err != nil { - return err - } - iNdEx = postIndex - default: - var sizeOfWire int - for { - sizeOfWire++ - wire >>= 7 - if wire == 0 { - break - } - } - iNdEx -= sizeOfWire - skippy, err := skipFirestormRows(data[iNdEx:]) - if err != nil { - return err - } - if skippy < 0 { - return ErrInvalidLengthFirestormRows - } - if (iNdEx + skippy) > l { - return io.ErrUnexpectedEOF - } - m.XXX_unrecognized = append(m.XXX_unrecognized, data[iNdEx:iNdEx+skippy]...) - iNdEx += skippy - } - } - if hasFields[0]&uint64(0x00000001) == 0 { - return new(github_com_golang_protobuf_proto.RequiredNotSetError) - } - - return nil -} -func (m *StoredValue) Unmarshal(data []byte) error { - l := len(data) - iNdEx := 0 - for iNdEx < l { - var wire uint64 - for shift := uint(0); ; shift += 7 { - if iNdEx >= l { - return io.ErrUnexpectedEOF - } - b := data[iNdEx] - iNdEx++ - wire |= (uint64(b) & 0x7F) << shift - if b < 0x80 { - break - } - } - fieldNum := int32(wire >> 3) - wireType := int(wire & 0x7) - switch fieldNum { - case 1: - if wireType != 2 { - return fmt.Errorf("proto: wrong wireType = %d for field Raw", wireType) - } - var byteLen int - for shift := uint(0); ; shift += 7 { - if iNdEx >= l { - return io.ErrUnexpectedEOF - } - b := data[iNdEx] - iNdEx++ - byteLen |= (int(b) & 0x7F) << shift - if b < 0x80 { - break - } - } - if byteLen < 0 { - return ErrInvalidLengthFirestormRows - } - postIndex := iNdEx + byteLen - if postIndex > l { - return io.ErrUnexpectedEOF - } - m.Raw = append([]byte{}, data[iNdEx:postIndex]...) - iNdEx = postIndex - default: - var sizeOfWire int - for { - sizeOfWire++ - wire >>= 7 - if wire == 0 { - break - } - } - iNdEx -= sizeOfWire - skippy, err := skipFirestormRows(data[iNdEx:]) - if err != nil { - return err - } - if skippy < 0 { - return ErrInvalidLengthFirestormRows - } - if (iNdEx + skippy) > l { - return io.ErrUnexpectedEOF - } - m.XXX_unrecognized = append(m.XXX_unrecognized, data[iNdEx:iNdEx+skippy]...) - iNdEx += skippy - } - } - - return nil -} -func skipFirestormRows(data []byte) (n int, err error) { - l := len(data) - iNdEx := 0 - for iNdEx < l { - var wire uint64 - for shift := uint(0); ; shift += 7 { - if iNdEx >= l { - return 0, io.ErrUnexpectedEOF - } - b := data[iNdEx] - iNdEx++ - wire |= (uint64(b) & 0x7F) << shift - if b < 0x80 { - break - } - } - wireType := int(wire & 0x7) - switch wireType { - case 0: - for { - if iNdEx >= l { - return 0, io.ErrUnexpectedEOF - } - iNdEx++ - if data[iNdEx-1] < 0x80 { - break - } - } - return iNdEx, nil - case 1: - iNdEx += 8 - return iNdEx, nil - case 2: - var length int - for shift := uint(0); ; shift += 7 { - if iNdEx >= l { - return 0, io.ErrUnexpectedEOF - } - b := data[iNdEx] - iNdEx++ - length |= (int(b) & 0x7F) << shift - if b < 0x80 { - break - } - } - iNdEx += length - if length < 0 { - return 0, ErrInvalidLengthFirestormRows - } - return iNdEx, nil - case 3: - for { - var innerWire uint64 - var start int = iNdEx - for shift := uint(0); ; shift += 7 { - if iNdEx >= l { - return 0, io.ErrUnexpectedEOF - } - b := data[iNdEx] - iNdEx++ - innerWire |= (uint64(b) & 0x7F) << shift - if b < 0x80 { - break - } - } - innerWireType := int(innerWire & 0x7) - if innerWireType == 4 { - break - } - next, err := skipFirestormRows(data[start:]) - if err != nil { - return 0, err - } - iNdEx = start + next - } - return iNdEx, nil - case 4: - return iNdEx, nil - case 5: - iNdEx += 4 - return iNdEx, nil - default: - return 0, fmt.Errorf("proto: illegal wireType %d", wireType) - } - } - panic("unreachable") -} - -var ( - ErrInvalidLengthFirestormRows = fmt.Errorf("proto: negative length found during unmarshaling") -) - -func (m *VersionValue) Size() (n int) { - var l int - _ = l - if m.Version != nil { - n += 1 + sovFirestormRows(uint64(*m.Version)) - } - if m.XXX_unrecognized != nil { - n += len(m.XXX_unrecognized) - } - return n -} - -func (m *FieldValue) Size() (n int) { - var l int - _ = l - if m.Name != nil { - l = len(*m.Name) - n += 1 + l + sovFirestormRows(uint64(l)) - } - if m.XXX_unrecognized != nil { - n += len(m.XXX_unrecognized) - } - return n -} - -func (m *DictionaryValue) Size() (n int) { - var l int - _ = l - if m.Count != nil { - n += 1 + sovFirestormRows(uint64(*m.Count)) - } - if m.XXX_unrecognized != nil { - n += len(m.XXX_unrecognized) - } - return n -} - -func (m *TermVector) Size() (n int) { - var l int - _ = l - if m.Field != nil { - n += 1 + sovFirestormRows(uint64(*m.Field)) - } - if m.Pos != nil { - n += 1 + sovFirestormRows(uint64(*m.Pos)) - } - if m.Start != nil { - n += 1 + sovFirestormRows(uint64(*m.Start)) - } - if m.End != nil { - n += 1 + sovFirestormRows(uint64(*m.End)) - } - if len(m.ArrayPositions) > 0 { - for _, e := range m.ArrayPositions { - n += 1 + sovFirestormRows(uint64(e)) - } - } - if m.XXX_unrecognized != nil { - n += len(m.XXX_unrecognized) - } - return n -} - -func (m *TermFreqValue) Size() (n int) { - var l int - _ = l - if m.Freq != nil { - n += 1 + sovFirestormRows(uint64(*m.Freq)) - } - if m.Norm != nil { - n += 5 - } - if len(m.Vectors) > 0 { - for _, e := range m.Vectors { - l = e.Size() - n += 1 + l + sovFirestormRows(uint64(l)) - } - } - if m.XXX_unrecognized != nil { - n += len(m.XXX_unrecognized) - } - return n -} - -func (m *StoredValue) Size() (n int) { - var l int - _ = l - if m.Raw != nil { - l = len(m.Raw) - n += 1 + l + sovFirestormRows(uint64(l)) - } - if m.XXX_unrecognized != nil { - n += len(m.XXX_unrecognized) - } - return n -} - -func sovFirestormRows(x uint64) (n int) { - for { - n++ - x >>= 7 - if x == 0 { - break - } - } - return n -} -func sozFirestormRows(x uint64) (n int) { - return sovFirestormRows(uint64((x << 1) ^ uint64((int64(x) >> 63)))) -} -func (m *VersionValue) Marshal() (data []byte, err error) { - size := m.Size() - data = make([]byte, size) - n, err := m.MarshalTo(data) - if err != nil { - return nil, err - } - return data[:n], nil -} - -func (m *VersionValue) MarshalTo(data []byte) (n int, err error) { - var i int - _ = i - var l int - _ = l - if m.Version == nil { - return 0, new(github_com_golang_protobuf_proto.RequiredNotSetError) - } else { - data[i] = 0x8 - i++ - i = encodeVarintFirestormRows(data, i, uint64(*m.Version)) - } - if m.XXX_unrecognized != nil { - i += copy(data[i:], m.XXX_unrecognized) - } - return i, nil -} - -func (m *FieldValue) Marshal() (data []byte, err error) { - size := m.Size() - data = make([]byte, size) - n, err := m.MarshalTo(data) - if err != nil { - return nil, err - } - return data[:n], nil -} - -func (m *FieldValue) MarshalTo(data []byte) (n int, err error) { - var i int - _ = i - var l int - _ = l - if m.Name == nil { - return 0, new(github_com_golang_protobuf_proto.RequiredNotSetError) - } else { - data[i] = 0xa - i++ - i = encodeVarintFirestormRows(data, i, uint64(len(*m.Name))) - i += copy(data[i:], *m.Name) - } - if m.XXX_unrecognized != nil { - i += copy(data[i:], m.XXX_unrecognized) - } - return i, nil -} - -func (m *DictionaryValue) Marshal() (data []byte, err error) { - size := m.Size() - data = make([]byte, size) - n, err := m.MarshalTo(data) - if err != nil { - return nil, err - } - return data[:n], nil -} - -func (m *DictionaryValue) MarshalTo(data []byte) (n int, err error) { - var i int - _ = i - var l int - _ = l - if m.Count != nil { - data[i] = 0x8 - i++ - i = encodeVarintFirestormRows(data, i, uint64(*m.Count)) - } - if m.XXX_unrecognized != nil { - i += copy(data[i:], m.XXX_unrecognized) - } - return i, nil -} - -func (m *TermVector) Marshal() (data []byte, err error) { - size := m.Size() - data = make([]byte, size) - n, err := m.MarshalTo(data) - if err != nil { - return nil, err - } - return data[:n], nil -} - -func (m *TermVector) MarshalTo(data []byte) (n int, err error) { - var i int - _ = i - var l int - _ = l - if m.Field != nil { - data[i] = 0x8 - i++ - i = encodeVarintFirestormRows(data, i, uint64(*m.Field)) - } - if m.Pos != nil { - data[i] = 0x10 - i++ - i = encodeVarintFirestormRows(data, i, uint64(*m.Pos)) - } - if m.Start != nil { - data[i] = 0x18 - i++ - i = encodeVarintFirestormRows(data, i, uint64(*m.Start)) - } - if m.End != nil { - data[i] = 0x20 - i++ - i = encodeVarintFirestormRows(data, i, uint64(*m.End)) - } - if len(m.ArrayPositions) > 0 { - for _, num := range m.ArrayPositions { - data[i] = 0x28 - i++ - i = encodeVarintFirestormRows(data, i, uint64(num)) - } - } - if m.XXX_unrecognized != nil { - i += copy(data[i:], m.XXX_unrecognized) - } - return i, nil -} - -func (m *TermFreqValue) Marshal() (data []byte, err error) { - size := m.Size() - data = make([]byte, size) - n, err := m.MarshalTo(data) - if err != nil { - return nil, err - } - return data[:n], nil -} - -func (m *TermFreqValue) MarshalTo(data []byte) (n int, err error) { - var i int - _ = i - var l int - _ = l - if m.Freq == nil { - return 0, new(github_com_golang_protobuf_proto.RequiredNotSetError) - } else { - data[i] = 0x8 - i++ - i = encodeVarintFirestormRows(data, i, uint64(*m.Freq)) - } - if m.Norm != nil { - data[i] = 0x15 - i++ - i = encodeFixed32FirestormRows(data, i, uint32(math.Float32bits(*m.Norm))) - } - if len(m.Vectors) > 0 { - for _, msg := range m.Vectors { - data[i] = 0x1a - i++ - i = encodeVarintFirestormRows(data, i, uint64(msg.Size())) - n, err := msg.MarshalTo(data[i:]) - if err != nil { - return 0, err - } - i += n - } - } - if m.XXX_unrecognized != nil { - i += copy(data[i:], m.XXX_unrecognized) - } - return i, nil -} - -func (m *StoredValue) Marshal() (data []byte, err error) { - size := m.Size() - data = make([]byte, size) - n, err := m.MarshalTo(data) - if err != nil { - return nil, err - } - return data[:n], nil -} - -func (m *StoredValue) MarshalTo(data []byte) (n int, err error) { - var i int - _ = i - var l int - _ = l - if m.Raw != nil { - data[i] = 0xa - i++ - i = encodeVarintFirestormRows(data, i, uint64(len(m.Raw))) - i += copy(data[i:], m.Raw) - } - if m.XXX_unrecognized != nil { - i += copy(data[i:], m.XXX_unrecognized) - } - return i, nil -} - -func encodeFixed64FirestormRows(data []byte, offset int, v uint64) int { - data[offset] = uint8(v) - data[offset+1] = uint8(v >> 8) - data[offset+2] = uint8(v >> 16) - data[offset+3] = uint8(v >> 24) - data[offset+4] = uint8(v >> 32) - data[offset+5] = uint8(v >> 40) - data[offset+6] = uint8(v >> 48) - data[offset+7] = uint8(v >> 56) - return offset + 8 -} -func encodeFixed32FirestormRows(data []byte, offset int, v uint32) int { - data[offset] = uint8(v) - data[offset+1] = uint8(v >> 8) - data[offset+2] = uint8(v >> 16) - data[offset+3] = uint8(v >> 24) - return offset + 4 -} -func encodeVarintFirestormRows(data []byte, offset int, v uint64) int { - for v >= 1<<7 { - data[offset] = uint8(v&0x7f | 0x80) - v >>= 7 - offset++ - } - data[offset] = uint8(v) - return offset + 1 -} diff --git a/vendor/github.com/blevesearch/bleve/index/firestorm/firestorm_rows.proto b/vendor/github.com/blevesearch/bleve/index/firestorm/firestorm_rows.proto deleted file mode 100644 index 9a69c76..0000000 --- a/vendor/github.com/blevesearch/bleve/index/firestorm/firestorm_rows.proto +++ /dev/null @@ -1,31 +0,0 @@ -package firestorm; - -message VersionValue { - required uint64 version = 1; -} - -message FieldValue { - required string name = 1; -} - -message DictionaryValue { - optional uint64 count = 1; // number of documents using this term in this field -} - -message TermVector { - optional uint32 field = 1; // field optional if redundant, required for composite fields - optional uint64 pos = 2; // positional offset within the field - optional uint64 start = 3; // start byte offset - optional uint64 end = 4; // end byte offset - repeated uint64 arrayPositions = 5; // array positions -} - -message TermFreqValue { - required uint64 freq = 1; // frequency of the term occurance within this field - optional float norm = 2; // normalization factor - repeated TermVector vectors = 3; // term vectors -} - -message StoredValue { - optional bytes raw = 1; // raw bytes -} \ No newline at end of file diff --git a/vendor/github.com/blevesearch/bleve/index/firestorm/garbage.go b/vendor/github.com/blevesearch/bleve/index/firestorm/garbage.go deleted file mode 100644 index 7a2abe4..0000000 --- a/vendor/github.com/blevesearch/bleve/index/firestorm/garbage.go +++ /dev/null @@ -1,235 +0,0 @@ -// Copyright (c) 2015 Couchbase, Inc. -// Licensed under the Apache License, Version 2.0 (the "License"); you may not use this file -// except in compliance with the License. You may obtain a copy of the License at -// http://www.apache.org/licenses/LICENSE-2.0 -// Unless required by applicable law or agreed to in writing, software distributed under the -// License is distributed on an "AS IS" BASIS, WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, -// either express or implied. See the License for the specific language governing permissions -// and limitations under the License. - -package firestorm - -import ( - "math" - "sync" - "time" -) - -const DefaultGarbageThreshold = 10 -const DefaultMaxDocsPerPass = 1000 - -var DefaultGarbageSleep = 15 * time.Second - -type GarbageCollector struct { - f *Firestorm - garbageThreshold int - garbageSleep time.Duration - maxDocsPerPass int - quit chan struct{} - - mutex sync.RWMutex - workingSet map[uint64][]byte - closeWait sync.WaitGroup -} - -func NewGarbageCollector(f *Firestorm) *GarbageCollector { - rv := GarbageCollector{ - f: f, - garbageThreshold: DefaultGarbageThreshold, - garbageSleep: DefaultGarbageSleep, - maxDocsPerPass: DefaultMaxDocsPerPass, - quit: make(chan struct{}), - workingSet: make(map[uint64][]byte), - } - return &rv -} - -func (gc *GarbageCollector) Notify(docNum uint64, docId []byte) { - gc.mutex.Lock() - defer gc.mutex.Unlock() - gc.workingSet[docNum] = docId -} - -func (gc *GarbageCollector) Start() { - gc.closeWait.Add(1) - go gc.run() -} - -func (gc *GarbageCollector) Stop() { - close(gc.quit) - gc.closeWait.Wait() -} - -func (gc *GarbageCollector) run() { - tick := time.Tick(gc.garbageSleep) - for { - select { - case <-gc.quit: - logger.Printf("garbage collector asked to quit") - gc.closeWait.Done() - return - case <-tick: - logger.Printf("garbage collector ticked") - garbageSize := gc.f.compensator.GarbageCount() - docSize, err := gc.f.DocCount() - if err != nil { - logger.Printf("garbage collector error getting doc count: %v", err) - continue - } - if docSize == 0 { - continue - } - garbageRatio := int(uint64(garbageSize) / docSize) - if garbageRatio > gc.garbageThreshold { - gc.cleanup() - } else { - logger.Printf("garbage ratio only %d, waiting", garbageRatio) - } - - } - } -} - -func (gc *GarbageCollector) NextBatch(n int) []uint64 { - gc.mutex.RLock() - defer gc.mutex.RUnlock() - - rv := make([]uint64, 0, n) - i := 0 - for k := range gc.workingSet { - rv = append(rv, k) - i++ - if i > n { - break - } - } - - return rv -} - -func (gc *GarbageCollector) cleanup() { - logger.Printf("garbage collector starting") - // get list of deleted doc numbers to work on this pass - deletedDocNumsList := gc.NextBatch(gc.maxDocsPerPass) //gc.f.deletedDocNumbers.Keys(gc.maxDocsPerPass) - logger.Printf("found %d doc numbers to cleanup", len(deletedDocNumsList)) - - // put these documents numbers in a map, for faster checking - // and for organized keys to be deleted - deletedDocNums := make(map[uint64][][]byte) - for _, deletedDocNum := range deletedDocNumsList { - deletedDocNums[deletedDocNum] = make([][]byte, 0) - } - - reader, err := gc.f.store.Reader() - if err != nil { - logger.Printf("garbage collector fatal: %v", err) - return - } - defer func() { - if cerr := reader.Close(); err == nil && cerr != nil { - err = cerr - } - }() - - // walk all the term freq rows (where field > 0) - termFreqStart := TermFreqIteratorStart(0, []byte{ByteSeparator}) - termFreqEnd := TermFreqIteratorStart(math.MaxUint16, []byte{ByteSeparator}) - - var tfr TermFreqRow - dictionaryDeltas := make(map[string]int64) - err = visitRange(reader, termFreqStart, termFreqEnd, func(key, val []byte) (bool, error) { - err := tfr.ParseKey(key) - if err != nil { - return false, err - } - docNum := tfr.DocNum() - if docNumKeys, deleted := deletedDocNums[docNum]; deleted { - // this doc number has been deleted, place key into map - deletedDocNums[docNum] = append(docNumKeys, key) - if tfr.Field() != 0 { - drk := tfr.DictionaryRowKey() - dictionaryDeltas[string(drk)] -= 1 - } - } - return true, nil - }) - if err != nil { - logger.Printf("garbage collector fatal: %v", err) - return - } - - // walk all the stored rows - var sr StoredRow - err = visitPrefix(reader, StoredKeyPrefix, func(key, val []byte) (bool, error) { - err := sr.ParseKey(key) - if err != nil { - return false, err - } - docNum := sr.DocNum() - if docNumKeys, deleted := deletedDocNums[docNum]; deleted { - // this doc number has been deleted, place key into map - deletedDocNums[docNum] = append(docNumKeys, key) - } - return true, nil - }) - if err != nil { - logger.Printf("garbage collector fatal: %v", err) - return - } - - // now process each doc one at a time - for docNum, docKeys := range deletedDocNums { - - // delete keys for a doc number - logger.Printf("deleting keys for %d", docNum) - // open a writer - writer, err := gc.f.store.Writer() - if err != nil { - _ = writer.Close() - logger.Printf("garbage collector fatal: %v", err) - return - } - - // prepare batch - wb := writer.NewBatch() - - for _, k := range docKeys { - wb.Delete(k) - } - - err = writer.ExecuteBatch(wb) - if err != nil { - _ = writer.Close() - logger.Printf("garbage collector fatal: %v", err) - return - } - logger.Printf("deleted %d keys", len(docKeys)) - - // remove it from delete keys list - docID := gc.workingSet[docNum] - delete(gc.workingSet, docNum) - gc.f.compensator.GarbageCollect([]uint64{docNum}) - - // now delete the original marker row (field 0) - tfidrow := NewTermFreqRow(0, nil, docID, docNum, 0, 0, nil) - markerRowKey := tfidrow.Key() - - markerBatch := writer.NewBatch() - markerBatch.Delete(markerRowKey) - err = writer.ExecuteBatch(markerBatch) - if err != nil { - logger.Printf("garbage collector fatal: %v", err) - return - } - err = writer.Close() - if err != nil { - logger.Printf("garbage collector fatal: %v", err) - return - } - } - - // updating dictionary in one batch - gc.f.dictUpdater.NotifyBatch(dictionaryDeltas) - - logger.Printf("garbage collector finished") -} diff --git a/vendor/github.com/blevesearch/bleve/index/firestorm/internal.go b/vendor/github.com/blevesearch/bleve/index/firestorm/internal.go deleted file mode 100644 index 8b7aa7a..0000000 --- a/vendor/github.com/blevesearch/bleve/index/firestorm/internal.go +++ /dev/null @@ -1,67 +0,0 @@ -// Copyright (c) 2015 Couchbase, Inc. -// Licensed under the Apache License, Version 2.0 (the "License"); you may not use this file -// except in compliance with the License. You may obtain a copy of the License at -// http://www.apache.org/licenses/LICENSE-2.0 -// Unless required by applicable law or agreed to in writing, software distributed under the -// License is distributed on an "AS IS" BASIS, WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, -// either express or implied. See the License for the specific language governing permissions -// and limitations under the License. - -package firestorm - -import "fmt" - -var InternalKeyPrefix = []byte{'i'} - -type InternalRow struct { - key []byte - val []byte -} - -func NewInternalRow(key, val []byte) *InternalRow { - rv := InternalRow{ - key: key, - val: val, - } - return &rv -} - -func NewInternalRowKV(key, value []byte) (*InternalRow, error) { - rv := InternalRow{} - rv.key = key[1:] - rv.val = value - return &rv, nil -} - -func (ir *InternalRow) KeySize() int { - return 1 + len(ir.key) -} - -func (ir *InternalRow) KeyTo(buf []byte) (int, error) { - buf[0] = 'i' - copy(buf[1:], ir.key) - return 1 + len(ir.key), nil -} - -func (ir *InternalRow) Key() []byte { - buf := make([]byte, ir.KeySize()) - n, _ := ir.KeyTo(buf) - return buf[:n] -} - -func (ir *InternalRow) ValueSize() int { - return len(ir.val) -} - -func (ir *InternalRow) ValueTo(buf []byte) (int, error) { - copy(buf, ir.val) - return len(ir.val), nil -} - -func (ir *InternalRow) Value() []byte { - return ir.val -} - -func (ir *InternalRow) String() string { - return fmt.Sprintf("InternalStore - Key: %s (% x) Val: %s (% x)", ir.key, ir.key, ir.val, ir.val) -} diff --git a/vendor/github.com/blevesearch/bleve/index/firestorm/lookup.go b/vendor/github.com/blevesearch/bleve/index/firestorm/lookup.go deleted file mode 100644 index fee6c82..0000000 --- a/vendor/github.com/blevesearch/bleve/index/firestorm/lookup.go +++ /dev/null @@ -1,146 +0,0 @@ -// Copyright (c) 2015 Couchbase, Inc. -// Licensed under the Apache License, Version 2.0 (the "License"); you may not use this file -// except in compliance with the License. You may obtain a copy of the License at -// http://www.apache.org/licenses/LICENSE-2.0 -// Unless required by applicable law or agreed to in writing, software distributed under the -// License is distributed on an "AS IS" BASIS, WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, -// either express or implied. See the License for the specific language governing permissions -// and limitations under the License. - -package firestorm - -import ( - "fmt" - "sync" - "sync/atomic" - "time" -) - -const channelBufferSize = 1000 - -type Lookuper struct { - tasksQueued uint64 - tasksDone uint64 - - f *Firestorm - workChan chan []*InFlightItem - quit chan struct{} - closeWait sync.WaitGroup -} - -func NewLookuper(f *Firestorm) *Lookuper { - rv := Lookuper{ - f: f, - workChan: make(chan []*InFlightItem, channelBufferSize), - quit: make(chan struct{}), - } - return &rv -} - -func (l *Lookuper) NotifyBatch(items []*InFlightItem) { - atomic.AddUint64(&l.tasksQueued, 1) - l.workChan <- items -} - -func (l *Lookuper) Start() { - l.closeWait.Add(1) - go l.run() -} - -func (l *Lookuper) Stop() { - close(l.quit) - l.closeWait.Wait() -} - -func (l *Lookuper) run() { - for { - - select { - case <-l.quit: - logger.Printf("lookuper asked to quit") - l.closeWait.Done() - return - case items, ok := <-l.workChan: - if !ok { - logger.Printf("lookuper work channel closed unexpectedly, stopping") - return - } - l.lookupItems(items) - } - } -} - -func (l *Lookuper) lookupItems(items []*InFlightItem) { - for _, item := range items { - l.lookup(item) - } - atomic.AddUint64(&l.tasksDone, 1) -} - -func (l *Lookuper) lookup(item *InFlightItem) { - reader, err := l.f.store.Reader() - if err != nil { - logger.Printf("lookuper fatal: %v", err) - return - } - defer func() { - if cerr := reader.Close(); err == nil && cerr != nil { - err = cerr - } - }() - - prefix := TermFreqPrefixFieldTermDocId(0, nil, item.docID) - logger.Printf("lookuper prefix - % x", prefix) - var tfk TermFreqRow - docNums := make(DocNumberList, 0) - err = visitPrefix(reader, prefix, func(key, val []byte) (bool, error) { - logger.Printf("lookuper sees key % x", key) - err := tfk.ParseKey(key) - if err != nil { - return false, err - } - docNum := tfk.DocNum() - docNums = append(docNums, docNum) - return true, nil - }) - if err != nil { - logger.Printf("lookuper fatal: %v", err) - return - } - oldDocNums := make(DocNumberList, 0, len(docNums)) - for _, docNum := range docNums { - if item.docNum == 0 || docNum < item.docNum { - oldDocNums = append(oldDocNums, docNum) - } - } - logger.Printf("lookup migrating '%s' - %d - oldDocNums: %v", item.docID, item.docNum, oldDocNums) - l.f.compensator.Migrate(item.docID, item.docNum, oldDocNums) - if len(oldDocNums) == 0 && item.docNum != 0 { - // this was an add, not an update - atomic.AddUint64(&l.f.docCount, 1) - } else if len(oldDocNums) > 0 && item.docNum == 0 { - // this was a delete (and it previously existed) - atomic.AddUint64(&l.f.docCount, ^uint64(0)) - } -} - -// this is not intended to be used publicly, only for unit tests -// which depend on consistency we no longer provide -func (l *Lookuper) waitTasksDone(d time.Duration) error { - timeout := time.After(d) - tick := time.Tick(100 * time.Millisecond) - for { - select { - // Got a timeout! fail with a timeout error - case <-timeout: - return fmt.Errorf("timeout") - // Got a tick, we should check on doSomething() - case <-tick: - queued := atomic.LoadUint64(&l.tasksQueued) - done := atomic.LoadUint64(&l.tasksDone) - if queued == done { - return nil - } - } - } -} diff --git a/vendor/github.com/blevesearch/bleve/index/firestorm/merge.go b/vendor/github.com/blevesearch/bleve/index/firestorm/merge.go deleted file mode 100644 index d065e61..0000000 --- a/vendor/github.com/blevesearch/bleve/index/firestorm/merge.go +++ /dev/null @@ -1,71 +0,0 @@ -// Copyright (c) 2015 Couchbase, Inc. -// Licensed under the Apache License, Version 2.0 (the "License"); you may not use this file -// except in compliance with the License. You may obtain a copy of the License at -// http://www.apache.org/licenses/LICENSE-2.0 -// Unless required by applicable law or agreed to in writing, software distributed under the -// License is distributed on an "AS IS" BASIS, WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, -// either express or implied. See the License for the specific language governing permissions -// and limitations under the License. - -package firestorm - -import ( - "encoding/binary" -) - -var mergeOperator firestormMerge - -var dictionaryTermIncr []byte -var dictionaryTermDecr []byte - -func init() { - dictionaryTermIncr = make([]byte, 8) - binary.LittleEndian.PutUint64(dictionaryTermIncr, uint64(1)) - dictionaryTermDecr = make([]byte, 8) - var negOne = int64(-1) - binary.LittleEndian.PutUint64(dictionaryTermDecr, uint64(negOne)) -} - -type firestormMerge struct{} - -func (m *firestormMerge) FullMerge(key, existingValue []byte, operands [][]byte) ([]byte, bool) { - // set up record based on key - dr, err := NewDictionaryRowK(key) - if err != nil { - return nil, false - } - if len(existingValue) > 0 { - // if existing value, parse it - err = dr.parseDictionaryV(existingValue) - if err != nil { - return nil, false - } - } - - // now process operands - for _, operand := range operands { - next := int64(binary.LittleEndian.Uint64(operand)) - if next < 0 && uint64(-next) > dr.Count() { - // subtracting next from existing would overflow - dr.SetCount(0) - } else if next < 0 { - dr.SetCount(dr.Count() - uint64(-next)) - } else { - dr.SetCount(dr.Count() + uint64(next)) - } - } - - return dr.Value(), true -} - -func (m *firestormMerge) PartialMerge(key, leftOperand, rightOperand []byte) ([]byte, bool) { - left := int64(binary.LittleEndian.Uint64(leftOperand)) - right := int64(binary.LittleEndian.Uint64(rightOperand)) - rv := make([]byte, 8) - binary.LittleEndian.PutUint64(rv, uint64(left+right)) - return rv, true -} - -func (m *firestormMerge) Name() string { - return "firestormMerge" -} diff --git a/vendor/github.com/blevesearch/bleve/index/firestorm/reader.go b/vendor/github.com/blevesearch/bleve/index/firestorm/reader.go deleted file mode 100644 index 29cc5f6..0000000 --- a/vendor/github.com/blevesearch/bleve/index/firestorm/reader.go +++ /dev/null @@ -1,220 +0,0 @@ -// Copyright (c) 2015 Couchbase, Inc. -// Licensed under the Apache License, Version 2.0 (the "License"); you may not use this file -// except in compliance with the License. You may obtain a copy of the License at -// http://www.apache.org/licenses/LICENSE-2.0 -// Unless required by applicable law or agreed to in writing, software distributed under the -// License is distributed on an "AS IS" BASIS, WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, -// either express or implied. See the License for the specific language governing permissions -// and limitations under the License. - -package firestorm - -import ( - "bytes" - "fmt" - "sort" - - "github.com/blevesearch/bleve/document" - "github.com/blevesearch/bleve/index" - "github.com/blevesearch/bleve/index/store" -) - -type firestormReader struct { - f *Firestorm - r store.KVReader - s *Snapshot - docCount uint64 -} - -func newFirestormReader(f *Firestorm) (index.IndexReader, error) { - r, err := f.store.Reader() - if err != nil { - return nil, fmt.Errorf("error opening store reader: %v", err) - } - docCount, err := f.DocCount() - if err != nil { - return nil, fmt.Errorf("error opening store reader: %v", err) - } - rv := firestormReader{ - f: f, - r: r, - s: f.compensator.Snapshot(), - docCount: docCount, - } - return &rv, nil -} - -func (r *firestormReader) TermFieldReader(term []byte, field string) (index.TermFieldReader, error) { - fieldIndex, fieldExists := r.f.fieldCache.FieldNamed(field, false) - if fieldExists { - return newFirestormTermFieldReader(r, uint16(fieldIndex), term) - } - return newFirestormTermFieldReader(r, ^uint16(0), []byte{ByteSeparator}) -} - -func (r *firestormReader) DocIDReader(start, end string) (index.DocIDReader, error) { - return newFirestormDocIDReader(r, start, end) -} - -func (r *firestormReader) FieldDict(field string) (index.FieldDict, error) { - return r.FieldDictRange(field, nil, nil) -} - -func (r *firestormReader) FieldDictRange(field string, startTerm []byte, endTerm []byte) (index.FieldDict, error) { - fieldIndex, fieldExists := r.f.fieldCache.FieldNamed(field, false) - if fieldExists { - return newFirestormDictionaryReader(r, uint16(fieldIndex), startTerm, endTerm) - } - return newFirestormDictionaryReader(r, ^uint16(0), []byte{ByteSeparator}, []byte{}) -} - -func (r *firestormReader) FieldDictPrefix(field string, termPrefix []byte) (index.FieldDict, error) { - return r.FieldDictRange(field, termPrefix, incrementBytes(termPrefix)) -} - -func (r *firestormReader) Document(id string) (*document.Document, error) { - docID := []byte(id) - docNum, err := r.currDocNumForId(docID) - if err != nil { - return nil, err - } else if docNum == 0 { - return nil, nil - } - rv := document.NewDocument(id) - prefix := StoredPrefixDocIDNum(docID, docNum) - err = visitPrefix(r.r, prefix, func(key, val []byte) (bool, error) { - safeVal := make([]byte, len(val)) - copy(safeVal, val) - row, err := NewStoredRowKV(key, safeVal) - if err != nil { - return false, err - } - if row != nil { - fieldName := r.f.fieldCache.FieldIndexed(row.field) - field := r.decodeFieldType(fieldName, row.arrayPositions, row.value.GetRaw()) - if field != nil { - rv.AddField(field) - } - } - return true, nil - }) - if err != nil { - return nil, err - } - return rv, nil -} - -func (r *firestormReader) decodeFieldType(name string, pos []uint64, value []byte) document.Field { - switch value[0] { - case 't': - return document.NewTextField(name, pos, value[1:]) - case 'n': - return document.NewNumericFieldFromBytes(name, pos, value[1:]) - case 'd': - return document.NewDateTimeFieldFromBytes(name, pos, value[1:]) - case 'b': - return document.NewBooleanFieldFromBytes(name, pos, value[1:]) - } - return nil -} - -func (r *firestormReader) currDocNumForId(docID []byte) (uint64, error) { - prefix := TermFreqPrefixFieldTermDocId(0, nil, docID) - docNums := make(DocNumberList, 0) - err := visitPrefix(r.r, prefix, func(key, val []byte) (bool, error) { - tfk, err := NewTermFreqRowKV(key, val) - if err != nil { - return false, err - } - docNum := tfk.DocNum() - docNums = append(docNums, docNum) - return true, nil - }) - if err != nil { - return 0, err - } - if len(docNums) > 0 { - sort.Sort(docNums) - return docNums[0], nil - } - return 0, nil -} - -func (r *firestormReader) DocumentFieldTerms(id string) (index.FieldTerms, error) { - - docID := []byte(id) - docNum, err := r.currDocNumForId(docID) - if err != nil { - return nil, err - } else if docNum == 0 { - return nil, nil - } - - rv := make(index.FieldTerms, 0) - // walk the term freqs - err = visitPrefix(r.r, TermFreqKeyPrefix, func(key, val []byte) (bool, error) { - tfr, err := NewTermFreqRowKV(key, val) - if err != nil { - return false, err - } - if bytes.Compare(tfr.DocID(), docID) == 0 && tfr.DocNum() == docNum && tfr.Field() != 0 { - fieldName := r.f.fieldCache.FieldIndexed(uint16(tfr.Field())) - terms, ok := rv[fieldName] - if !ok { - terms = make([]string, 0, 1) - } - terms = append(terms, string(tfr.Term())) - rv[fieldName] = terms - } - return true, nil - }) - if err != nil { - return nil, err - } - - return rv, nil -} - -func (r *firestormReader) Fields() ([]string, error) { - fields := make([]string, 0) - - err := visitPrefix(r.r, FieldKeyPrefix, func(key, val []byte) (bool, error) { - fieldRow, err := NewFieldRowKV(key, val) - if err != nil { - return false, err - } - fields = append(fields, fieldRow.Name()) - return true, nil - }) - if err != nil { - return nil, err - } - - return fields, nil -} - -func (r *firestormReader) GetInternal(key []byte) ([]byte, error) { - internalRow := NewInternalRow(key, nil) - return r.r.Get(internalRow.Key()) -} - -func (r *firestormReader) DocCount() uint64 { - return r.docCount -} - -func (r *firestormReader) Close() error { - return r.r.Close() -} - -func incrementBytes(in []byte) []byte { - rv := make([]byte, len(in)) - copy(rv, in) - for i := len(rv) - 1; i >= 0; i-- { - rv[i] = rv[i] + 1 - if rv[i] != 0 { - // didn't overflow, so stop - break - } - } - return rv -} diff --git a/vendor/github.com/blevesearch/bleve/index/firestorm/reader_dict.go b/vendor/github.com/blevesearch/bleve/index/firestorm/reader_dict.go deleted file mode 100644 index 0edeb3b..0000000 --- a/vendor/github.com/blevesearch/bleve/index/firestorm/reader_dict.go +++ /dev/null @@ -1,70 +0,0 @@ -// Copyright (c) 2015 Couchbase, Inc. -// Licensed under the Apache License, Version 2.0 (the "License"); you may not use this file -// except in compliance with the License. You may obtain a copy of the License at -// http://www.apache.org/licenses/LICENSE-2.0 -// Unless required by applicable law or agreed to in writing, software distributed under the -// License is distributed on an "AS IS" BASIS, WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, -// either express or implied. See the License for the specific language governing permissions -// and limitations under the License. - -package firestorm - -import ( - "fmt" - - "github.com/blevesearch/bleve/index" - "github.com/blevesearch/bleve/index/store" -) - -type firestormDictionaryReader struct { - r *firestormReader - field uint16 - start []byte - i store.KVIterator -} - -func newFirestormDictionaryReader(r *firestormReader, field uint16, start, end []byte) (*firestormDictionaryReader, error) { - startKey := DictionaryRowKey(field, start) - logger.Printf("start key '%s' - % x", startKey, startKey) - if end == nil { - end = []byte{ByteSeparator} - } - endKey := DictionaryRowKey(field, end) - logger.Printf("end key '%s' - % x", endKey, endKey) - i := r.r.RangeIterator(startKey, endKey) - rv := firestormDictionaryReader{ - r: r, - field: field, - start: startKey, - i: i, - } - return &rv, nil -} - -func (r *firestormDictionaryReader) Next() (*index.DictEntry, error) { - key, val, valid := r.i.Current() - if !valid { - return nil, nil - } - - logger.Printf("see key '%s' - % x", key, key) - - currRow, err := NewDictionaryRowKV(key, val) - if err != nil { - return nil, fmt.Errorf("unexpected error parsing dictionary row kv: %v", err) - } - rv := index.DictEntry{ - Term: string(currRow.term), - Count: currRow.Count(), - } - // advance the iterator to the next term - r.i.Next() - return &rv, nil -} - -func (r *firestormDictionaryReader) Close() error { - if r.i != nil { - return r.i.Close() - } - return nil -} diff --git a/vendor/github.com/blevesearch/bleve/index/firestorm/reader_docs.go b/vendor/github.com/blevesearch/bleve/index/firestorm/reader_docs.go deleted file mode 100644 index fcecaa7..0000000 --- a/vendor/github.com/blevesearch/bleve/index/firestorm/reader_docs.go +++ /dev/null @@ -1,120 +0,0 @@ -// Copyright (c) 2015 Couchbase, Inc. -// Licensed under the Apache License, Version 2.0 (the "License"); you may not use this file -// except in compliance with the License. You may obtain a copy of the License at -// http://www.apache.org/licenses/LICENSE-2.0 -// Unless required by applicable law or agreed to in writing, software distributed under the -// License is distributed on an "AS IS" BASIS, WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, -// either express or implied. See the License for the specific language governing permissions -// and limitations under the License. - -package firestorm - -import ( - "bytes" - - "github.com/blevesearch/bleve/index/store" -) - -type firestormDocIDReader struct { - r *firestormReader - start []byte - i store.KVIterator -} - -func newFirestormDocIDReader(r *firestormReader, start, end string) (*firestormDocIDReader, error) { - startKey := TermFreqIteratorStart(0, nil) - if start != "" { - startKey = TermFreqPrefixFieldTermDocId(0, nil, []byte(start)) - } - logger.Printf("start key '%s' - % x", startKey, startKey) - endKey := TermFreqIteratorStart(0, []byte{ByteSeparator}) - if end != "" { - endKey = TermFreqPrefixFieldTermDocId(0, nil, []byte(end)) - } - - logger.Printf("end key '%s' - % x", endKey, endKey) - - i := r.r.RangeIterator(startKey, endKey) - - rv := firestormDocIDReader{ - r: r, - start: startKey, - i: i, - } - - return &rv, nil -} - -func (r *firestormDocIDReader) Next() (string, error) { - if r.i != nil { - key, val, valid := r.i.Current() - for valid { - logger.Printf("see key: '%s' - % x", key, key) - tfrsByDocNum := make(map[uint64]*TermFreqRow) - tfr, err := NewTermFreqRowKV(key, val) - if err != nil { - return "", err - } - tfrsByDocNum[tfr.DocNum()] = tfr - - // now we have a possible row, but there may be more rows for the same docid - // find these now - err = r.findNextTfrsWithSameDocId(tfrsByDocNum, tfr.DocID()) - if err != nil { - return "", err - } - - docNumList := make(DocNumberList, 0, len(tfrsByDocNum)) - for dn := range tfrsByDocNum { - docNumList = append(docNumList, dn) - } - - logger.Printf("docNumList: %v", docNumList) - - highestValidDocNum := r.r.s.Which(tfr.docID, docNumList) - if highestValidDocNum == 0 { - // no valid doc number - key, val, valid = r.i.Current() - continue - } - logger.Printf("highest valid: %d", highestValidDocNum) - - tfr = tfrsByDocNum[highestValidDocNum] - return string(tfr.DocID()), nil - } - } - return "", nil -} - -// FIXME this is identical to the one in reader_terms.go -func (r *firestormDocIDReader) findNextTfrsWithSameDocId(tfrsByDocNum map[uint64]*TermFreqRow, docID []byte) error { - tfrDocIdPrefix := TermFreqPrefixFieldTermDocId(0, nil, docID) - r.i.Next() - key, val, valid := r.i.Current() - for valid && bytes.HasPrefix(key, tfrDocIdPrefix) { - tfr, err := NewTermFreqRowKV(key, val) - if err != nil { - return err - } - tfrsByDocNum[tfr.DocNum()] = tfr - r.i.Next() - key, val, valid = r.i.Current() - } - return nil -} - -func (r *firestormDocIDReader) Advance(docID string) (string, error) { - if r.i != nil { - tfrDocIdPrefix := TermFreqPrefixFieldTermDocId(0, nil, []byte(docID)) - r.i.Seek(tfrDocIdPrefix) - return r.Next() - } - return "", nil -} - -func (r *firestormDocIDReader) Close() error { - if r.i != nil { - return r.i.Close() - } - return nil -} diff --git a/vendor/github.com/blevesearch/bleve/index/firestorm/reader_terms.go b/vendor/github.com/blevesearch/bleve/index/firestorm/reader_terms.go deleted file mode 100644 index 9c1038f..0000000 --- a/vendor/github.com/blevesearch/bleve/index/firestorm/reader_terms.go +++ /dev/null @@ -1,162 +0,0 @@ -// Copyright (c) 2015 Couchbase, Inc. -// Licensed under the Apache License, Version 2.0 (the "License"); you may not use this file -// except in compliance with the License. You may obtain a copy of the License at -// http://www.apache.org/licenses/LICENSE-2.0 -// Unless required by applicable law or agreed to in writing, software distributed under the -// License is distributed on an "AS IS" BASIS, WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, -// either express or implied. See the License for the specific language governing permissions -// and limitations under the License. - -package firestorm - -import ( - "bytes" - "sync/atomic" - - "github.com/blevesearch/bleve/index" - "github.com/blevesearch/bleve/index/store" -) - -type firestormTermFieldReader struct { - r *firestormReader - field uint16 - term []byte - prefix []byte - count uint64 - i store.KVIterator -} - -func newFirestormTermFieldReader(r *firestormReader, field uint16, term []byte) (index.TermFieldReader, error) { - dictionaryKey := DictionaryRowKey(field, term) - dictionaryValue, err := r.r.Get(dictionaryKey) - if err != nil { - return nil, err - } - - prefix := TermFreqIteratorStart(field, term) - logger.Printf("starting term freq iterator at: '%s' - % x", prefix, prefix) - i := r.r.PrefixIterator(prefix) - rv := firestormTermFieldReader{ - r: r, - field: field, - term: term, - prefix: prefix, - i: i, - } - - // NOTE: in firestorm the dictionary row is advisory in nature - // it *may* tell us the correct out - // if this record does not exist, it DOES not mean that there is no - // usage, we must scan the term frequencies to be sure - if dictionaryValue != nil { - dictionaryRow, err := NewDictionaryRowKV(dictionaryKey, dictionaryValue) - if err != nil { - return nil, err - } - rv.count = dictionaryRow.Count() - } - - atomic.AddUint64(&r.f.stats.termSearchersStarted, uint64(1)) - return &rv, nil -} - -func (r *firestormTermFieldReader) Next() (*index.TermFieldDoc, error) { - if r.i != nil { - key, val, valid := r.i.Current() - for valid { - logger.Printf("see key: '%s' - % x", key, key) - tfrsByDocNum := make(map[uint64]*TermFreqRow) - tfr, err := NewTermFreqRowKV(key, val) - if err != nil { - return nil, err - } - tfrsByDocNum[tfr.DocNum()] = tfr - - // now we have a possible row, but there may be more rows for the same docid - // find these now - err = r.findNextTfrsWithSameDocId(tfrsByDocNum, tfr.DocID()) - if err != nil { - return nil, err - } - - docNumList := make(DocNumberList, 0, len(tfrsByDocNum)) - for dn := range tfrsByDocNum { - docNumList = append(docNumList, dn) - } - - logger.Printf("docNumList: %v", docNumList) - - highestValidDocNum := r.r.s.Which(tfr.docID, docNumList) - if highestValidDocNum == 0 { - // no valid doc number - key, val, valid = r.i.Current() - continue - } - logger.Printf("highest valid: %d", highestValidDocNum) - - tfr = tfrsByDocNum[highestValidDocNum] - - return &index.TermFieldDoc{ - ID: string(tfr.DocID()), - Freq: tfr.Freq(), - Norm: float64(tfr.Norm()), - Vectors: r.termFieldVectorsFromTermVectors(tfr.Vectors()), - }, nil - } - } - return nil, nil -} - -func (r *firestormTermFieldReader) findNextTfrsWithSameDocId(tfrsByDocNum map[uint64]*TermFreqRow, docID []byte) error { - tfrDocIdPrefix := TermFreqPrefixFieldTermDocId(r.field, r.term, docID) - r.i.Next() - key, val, valid := r.i.Current() - for valid && bytes.HasPrefix(key, tfrDocIdPrefix) { - tfr, err := NewTermFreqRowKV(key, val) - if err != nil { - return err - } - tfrsByDocNum[tfr.DocNum()] = tfr - r.i.Next() - key, val, valid = r.i.Current() - } - return nil -} - -func (r *firestormTermFieldReader) Advance(docID string) (*index.TermFieldDoc, error) { - if r.i != nil { - tfrDocIdPrefix := TermFreqPrefixFieldTermDocId(r.field, r.term, []byte(docID)) - r.i.Seek(tfrDocIdPrefix) - return r.Next() - } - return nil, nil -} - -func (r *firestormTermFieldReader) Count() uint64 { - return r.count -} - -func (r *firestormTermFieldReader) Close() error { - atomic.AddUint64(&r.r.f.stats.termSearchersFinished, uint64(1)) - if r.i != nil { - return r.i.Close() - } - return nil -} - -func (r *firestormTermFieldReader) termFieldVectorsFromTermVectors(in []*TermVector) []*index.TermFieldVector { - rv := make([]*index.TermFieldVector, len(in)) - - for i, tv := range in { - fieldName := r.r.f.fieldCache.FieldIndexed(uint16(tv.GetField())) - tfv := index.TermFieldVector{ - Field: fieldName, - ArrayPositions: tv.GetArrayPositions(), - Pos: tv.GetPos(), - Start: tv.GetStart(), - End: tv.GetEnd(), - } - rv[i] = &tfv - } - return rv -} diff --git a/vendor/github.com/blevesearch/bleve/index/firestorm/stats.go b/vendor/github.com/blevesearch/bleve/index/firestorm/stats.go deleted file mode 100644 index 0701e46..0000000 --- a/vendor/github.com/blevesearch/bleve/index/firestorm/stats.go +++ /dev/null @@ -1,51 +0,0 @@ -// Copyright (c) 2015 Couchbase, Inc. -// Licensed under the Apache License, Version 2.0 (the "License"); you may not use this file -// except in compliance with the License. You may obtain a copy of the License at -// http://www.apache.org/licenses/LICENSE-2.0 -// Unless required by applicable law or agreed to in writing, software distributed under the -// License is distributed on an "AS IS" BASIS, WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, -// either express or implied. See the License for the specific language governing permissions -// and limitations under the License. - -package firestorm - -import ( - "encoding/json" - "sync/atomic" - - "github.com/blevesearch/bleve/index/store" -) - -type indexStat struct { - updates, deletes, batches, errors uint64 - analysisTime, indexTime uint64 - termSearchersStarted uint64 - termSearchersFinished uint64 - numPlainTextBytesIndexed uint64 - f *Firestorm -} - -func (i *indexStat) statsMap() map[string]interface{} { - m := map[string]interface{}{} - m["updates"] = atomic.LoadUint64(&i.updates) - m["deletes"] = atomic.LoadUint64(&i.deletes) - m["batches"] = atomic.LoadUint64(&i.batches) - m["errors"] = atomic.LoadUint64(&i.errors) - m["analysis_time"] = atomic.LoadUint64(&i.analysisTime) - m["index_time"] = atomic.LoadUint64(&i.indexTime) - m["lookup_queue_len"] = len(i.f.lookuper.workChan) - m["term_searchers_started"] = atomic.LoadUint64(&i.termSearchersStarted) - m["term_searchers_finished"] = atomic.LoadUint64(&i.termSearchersFinished) - m["num_plain_text_bytes_indexed"] = atomic.LoadUint64(&i.numPlainTextBytesIndexed) - - if o, ok := i.f.store.(store.KVStoreStats); ok { - m["kv"] = o.StatsMap() - } - - return m -} - -func (i *indexStat) MarshalJSON() ([]byte, error) { - m := i.statsMap() - return json.Marshal(m) -} diff --git a/vendor/github.com/blevesearch/bleve/index/firestorm/stored.go b/vendor/github.com/blevesearch/bleve/index/firestorm/stored.go deleted file mode 100644 index a8a5917..0000000 --- a/vendor/github.com/blevesearch/bleve/index/firestorm/stored.go +++ /dev/null @@ -1,164 +0,0 @@ -// Copyright (c) 2015 Couchbase, Inc. -// Licensed under the Apache License, Version 2.0 (the "License"); you may not use this file -// except in compliance with the License. You may obtain a copy of the License at -// http://www.apache.org/licenses/LICENSE-2.0 -// Unless required by applicable law or agreed to in writing, software distributed under the -// License is distributed on an "AS IS" BASIS, WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, -// either express or implied. See the License for the specific language governing permissions -// and limitations under the License. - -package firestorm - -import ( - "bytes" - "encoding/binary" - "fmt" -) - -var StoredKeyPrefix = []byte{'s'} - -type StoredRow struct { - docID []byte - docNum uint64 - field uint16 - arrayPositions []uint64 - value StoredValue -} - -func NewStoredRow(docID []byte, docNum uint64, field uint16, arrayPositions []uint64, value []byte) *StoredRow { - rv := StoredRow{ - docID: docID, - docNum: docNum, - field: field, - arrayPositions: arrayPositions, - } - if len(arrayPositions) < 1 { - rv.arrayPositions = make([]uint64, 0) - } - rv.value.Raw = value // FIXME review do we need to copy? - return &rv -} - -func NewStoredRowKV(key, value []byte) (*StoredRow, error) { - rv := StoredRow{} - err := rv.ParseKey(key) - if err != nil { - return nil, err - } - err = rv.value.Unmarshal(value) - if err != nil { - return nil, err - } - return &rv, nil -} - -func (sr *StoredRow) ParseKey(key []byte) error { - buf := bytes.NewBuffer(key) - _, err := buf.ReadByte() // type - if err != nil { - return err - } - - sr.docID, err = buf.ReadBytes(ByteSeparator) - if len(sr.docID) < 2 { // 1 for min doc id length, 1 for separator - err = fmt.Errorf("invalid doc length 0") - return err - } - - sr.docID = sr.docID[:len(sr.docID)-1] // trim off separator byte - - sr.docNum, err = binary.ReadUvarint(buf) - if err != nil { - return err - } - - err = binary.Read(buf, binary.LittleEndian, &sr.field) - if err != nil { - return err - } - - sr.arrayPositions = make([]uint64, 0) - nextArrayPos, err := binary.ReadUvarint(buf) - for err == nil { - sr.arrayPositions = append(sr.arrayPositions, nextArrayPos) - nextArrayPos, err = binary.ReadUvarint(buf) - } - - return nil -} - -func (sr *StoredRow) KeySize() int { - return 1 + len(sr.docID) + 1 + binary.MaxVarintLen64 + 2 + (binary.MaxVarintLen64 * len(sr.arrayPositions)) -} - -func (sr *StoredRow) KeyTo(buf []byte) (int, error) { - buf[0] = 's' - copy(buf[1:], sr.docID) - buf[1+len(sr.docID)] = ByteSeparator - bytesUsed := 1 + len(sr.docID) + 1 - bytesUsed += binary.PutUvarint(buf[bytesUsed:], sr.docNum) - binary.LittleEndian.PutUint16(buf[bytesUsed:], sr.field) - bytesUsed += 2 - for _, arrayPosition := range sr.arrayPositions { - varbytes := binary.PutUvarint(buf[bytesUsed:], arrayPosition) - bytesUsed += varbytes - } - return bytesUsed, nil -} - -func (sr *StoredRow) Key() []byte { - - buf := make([]byte, sr.KeySize()) - n, _ := sr.KeyTo(buf) - return buf[:n] -} - -func (sr *StoredRow) ValueSize() int { - return sr.value.Size() -} - -func (sr *StoredRow) ValueTo(buf []byte) (int, error) { - return sr.value.MarshalTo(buf) -} - -func (sr *StoredRow) Value() []byte { - buf := make([]byte, sr.ValueSize()) - n, _ := sr.ValueTo(buf) - return buf[:n] -} - -func (sr *StoredRow) DocID() []byte { - return sr.docID -} - -func (sr *StoredRow) DocNum() uint64 { - return sr.docNum -} - -func (sr *StoredRow) String() string { - return fmt.Sprintf("StoredRow - Field: %d\n", sr.field) + - fmt.Sprintf("DocID '%s' - % x\n", sr.docID, sr.docID) + - fmt.Sprintf("DocNum %d\n", sr.docNum) + - fmt.Sprintf("Array Positions:\n%v", sr.arrayPositions) + - fmt.Sprintf("Value: % x", sr.value.GetRaw()) -} - -func StoredIteratorStartDocID(docID []byte) []byte { - docLen := len(docID) - buf := make([]byte, 1+docLen+1) - buf[0] = 's' - copy(buf[1:], docID) - buf[1+docLen] = ByteSeparator - return buf -} - -func StoredPrefixDocIDNum(docID []byte, docNum uint64) []byte { - docLen := len(docID) - buf := make([]byte, 1+docLen+1+binary.MaxVarintLen64) - buf[0] = 's' - copy(buf[1:], docID) - buf[1+docLen] = ByteSeparator - bytesUsed := 1 + docLen + 1 - bytesUsed += binary.PutUvarint(buf[bytesUsed:], docNum) - return buf[0:bytesUsed] -} diff --git a/vendor/github.com/blevesearch/bleve/index/firestorm/termfreq.go b/vendor/github.com/blevesearch/bleve/index/firestorm/termfreq.go deleted file mode 100644 index 6ba6078..0000000 --- a/vendor/github.com/blevesearch/bleve/index/firestorm/termfreq.go +++ /dev/null @@ -1,209 +0,0 @@ -// Copyright (c) 2015 Couchbase, Inc. -// Licensed under the Apache License, Version 2.0 (the "License"); you may not use this file -// except in compliance with the License. You may obtain a copy of the License at -// http://www.apache.org/licenses/LICENSE-2.0 -// Unless required by applicable law or agreed to in writing, software distributed under the -// License is distributed on an "AS IS" BASIS, WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, -// either express or implied. See the License for the specific language governing permissions -// and limitations under the License. - -package firestorm - -import ( - "bytes" - "encoding/binary" - "fmt" - - "github.com/golang/protobuf/proto" -) - -var TermFreqKeyPrefix = []byte{'t'} - -type TermFreqRow struct { - field uint16 - term []byte - docID []byte - docNum uint64 - value TermFreqValue -} - -func NewTermVector(field uint16, pos uint64, start uint64, end uint64, arrayPos []uint64) *TermVector { - rv := TermVector{} - - rv.Field = proto.Uint32(uint32(field)) - rv.Pos = proto.Uint64(pos) - rv.Start = proto.Uint64(start) - rv.End = proto.Uint64(end) - - if len(arrayPos) > 0 { - rv.ArrayPositions = make([]uint64, len(arrayPos)) - for i, apv := range arrayPos { - rv.ArrayPositions[i] = apv - } - } - - return &rv -} - -func NewTermFreqRow(field uint16, term []byte, docID []byte, docNum uint64, freq uint64, norm float32, termVectors []*TermVector) *TermFreqRow { - return InitTermFreqRow(&TermFreqRow{}, field, term, docID, docNum, freq, norm, termVectors) -} - -func InitTermFreqRow(tfr *TermFreqRow, field uint16, term []byte, docID []byte, docNum uint64, freq uint64, norm float32, termVectors []*TermVector) *TermFreqRow { - tfr.field = field - tfr.term = term - tfr.docID = docID - tfr.docNum = docNum - tfr.value.Freq = proto.Uint64(freq) - tfr.value.Norm = proto.Float32(norm) - tfr.value.Vectors = termVectors - return tfr -} - -func NewTermFreqRowKV(key, value []byte) (*TermFreqRow, error) { - rv := TermFreqRow{} - err := rv.ParseKey(key) - if err != nil { - return nil, err - } - err = rv.value.Unmarshal(value) - if err != nil { - return nil, err - } - return &rv, nil -} - -func (tfr *TermFreqRow) ParseKey(key []byte) error { - keyLen := len(key) - if keyLen < 3 { - return fmt.Errorf("invalid term frequency key, no valid field") - } - tfr.field = binary.LittleEndian.Uint16(key[1:3]) - - termStartPos := 3 - termEndPos := bytes.IndexByte(key[termStartPos:], ByteSeparator) - if termEndPos < 0 { - return fmt.Errorf("invalid term frequency key, no byte separator terminating term") - } - tfr.term = key[termStartPos : termStartPos+termEndPos] - - docStartPos := termStartPos + termEndPos + 1 - docEndPos := bytes.IndexByte(key[docStartPos:], ByteSeparator) - tfr.docID = key[docStartPos : docStartPos+docEndPos] - - docNumPos := docStartPos + docEndPos + 1 - tfr.docNum, _ = binary.Uvarint(key[docNumPos:]) - - return nil -} - -func (tfr *TermFreqRow) KeySize() int { - return 3 + len(tfr.term) + 1 + len(tfr.docID) + 1 + binary.MaxVarintLen64 -} - -func (tfr *TermFreqRow) KeyTo(buf []byte) (int, error) { - buf[0] = 't' - binary.LittleEndian.PutUint16(buf[1:3], tfr.field) - termLen := copy(buf[3:], tfr.term) - buf[3+termLen] = ByteSeparator - docLen := copy(buf[3+termLen+1:], tfr.docID) - buf[3+termLen+1+docLen] = ByteSeparator - used := binary.PutUvarint(buf[3+termLen+1+docLen+1:], tfr.docNum) - return 3 + termLen + 1 + docLen + 1 + used, nil -} - -func (tfr *TermFreqRow) Key() []byte { - buf := make([]byte, tfr.KeySize()) - n, _ := tfr.KeyTo(buf) - return buf[:n] -} - -func (tfr *TermFreqRow) ValueSize() int { - return tfr.value.Size() -} - -func (tfr *TermFreqRow) ValueTo(buf []byte) (int, error) { - return tfr.value.MarshalTo(buf) -} - -func (tfr *TermFreqRow) Value() []byte { - buf := make([]byte, tfr.ValueSize()) - n, _ := tfr.ValueTo(buf) - return buf[:n] -} - -func (tfr *TermFreqRow) String() string { - vectors := "" - for i, v := range tfr.value.GetVectors() { - vectors += fmt.Sprintf("%d - Field: %d Pos: %d Start: %d End: %d ArrayPos: %v - %#v\n", i, v.GetField(), v.GetPos(), v.GetStart(), v.GetEnd(), v.GetArrayPositions(), v.ArrayPositions) - } - return fmt.Sprintf("TermFreqRow - Field: %d\n", tfr.field) + - fmt.Sprintf("Term '%s' - % x\n", tfr.term, tfr.term) + - fmt.Sprintf("DocID '%s' - % x\n", tfr.docID, tfr.docID) + - fmt.Sprintf("DocNum %d\n", tfr.docNum) + - fmt.Sprintf("Freq: %d\n", tfr.value.GetFreq()) + - fmt.Sprintf("Norm: %f\n", tfr.value.GetNorm()) + - fmt.Sprintf("Vectors:\n%s", vectors) -} - -func (tfr *TermFreqRow) Field() uint16 { - return tfr.field -} - -func (tfr *TermFreqRow) Term() []byte { - return tfr.term -} - -func (tfr *TermFreqRow) DocID() []byte { - return tfr.docID -} - -func (tfr *TermFreqRow) DocNum() uint64 { - return tfr.docNum -} - -func (tfr *TermFreqRow) Norm() float32 { - return tfr.value.GetNorm() -} - -func (tfr *TermFreqRow) Freq() uint64 { - return tfr.value.GetFreq() -} - -func (tfr *TermFreqRow) Vectors() []*TermVector { - return tfr.value.GetVectors() -} - -func (tfr *TermFreqRow) DictionaryRowKeySize() int { - return 3 + len(tfr.term) -} - -func (tfr *TermFreqRow) DictionaryRowKeyTo(buf []byte) (int, error) { - dr := NewDictionaryRow(tfr.field, tfr.term, 0) - return dr.KeyTo(buf) -} - -func (tfr *TermFreqRow) DictionaryRowKey() []byte { - dr := NewDictionaryRow(tfr.field, tfr.term, 0) - return dr.Key() -} - -func TermFreqIteratorStart(field uint16, term []byte) []byte { - buf := make([]byte, 3+len(term)+1) - buf[0] = 't' - binary.LittleEndian.PutUint16(buf[1:3], field) - termLen := copy(buf[3:], term) - buf[3+termLen] = ByteSeparator - return buf -} - -func TermFreqPrefixFieldTermDocId(field uint16, term []byte, docID []byte) []byte { - buf := make([]byte, 3+len(term)+1+len(docID)+1) - buf[0] = 't' - binary.LittleEndian.PutUint16(buf[1:3], field) - termLen := copy(buf[3:], term) - buf[3+termLen] = ByteSeparator - docLen := copy(buf[3+termLen+1:], docID) - buf[3+termLen+1+docLen] = ByteSeparator - return buf -} diff --git a/vendor/github.com/blevesearch/bleve/index/firestorm/util.go b/vendor/github.com/blevesearch/bleve/index/firestorm/util.go deleted file mode 100644 index b52dadc..0000000 --- a/vendor/github.com/blevesearch/bleve/index/firestorm/util.go +++ /dev/null @@ -1,99 +0,0 @@ -// Copyright (c) 2015 Couchbase, Inc. -// Licensed under the Apache License, Version 2.0 (the "License"); you may not use this file -// except in compliance with the License. You may obtain a copy of the License at -// http://www.apache.org/licenses/LICENSE-2.0 -// Unless required by applicable law or agreed to in writing, software distributed under the -// License is distributed on an "AS IS" BASIS, WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, -// either express or implied. See the License for the specific language governing permissions -// and limitations under the License. - -package firestorm - -import ( - "io/ioutil" - "log" - - "github.com/blevesearch/bleve/index/store" -) - -type KVVisitor func(key, val []byte) (bool, error) - -func visitPrefix(reader store.KVReader, prefix []byte, visitor KVVisitor) (err error) { - start := prefix - if start == nil { - start = []byte{} - } - it := reader.PrefixIterator(start) - defer func() { - if cerr := it.Close(); err == nil && cerr != nil { - err = cerr - } - }() - k, v, valid := it.Current() - for valid { - var cont bool - cont, err = visitor(k, v) - if err != nil { - // visitor encountered an error, stop and return it - return - } - if !cont { - // vistor has requested we stop iteration, return nil - return - } - it.Next() - k, v, valid = it.Current() - } - return -} - -func visitRange(reader store.KVReader, start, end []byte, visitor KVVisitor) (err error) { - it := reader.RangeIterator(start, end) - defer func() { - if cerr := it.Close(); err == nil && cerr != nil { - err = cerr - } - }() - k, v, valid := it.Current() - for valid { - var cont bool - cont, err = visitor(k, v) - if err != nil { - // visitor encountered an error, stop and return it - return - } - if !cont { - // vistor has requested we stop iteration, return nil - return - } - it.Next() - k, v, valid = it.Current() - } - return -} - -type DocNumberList []uint64 - -func (l DocNumberList) Len() int { return len(l) } -func (l DocNumberList) Less(i, j int) bool { return l[i] > l[j] } -func (l DocNumberList) Swap(i, j int) { l[i], l[j] = l[j], l[i] } - -// HighestValid returns the highest valid doc number -// from a *SORTED* DocNumberList -// if no doc number in the list is valid, then 0 -func (l DocNumberList) HighestValid(maxRead uint64) uint64 { - for _, dn := range l { - if dn <= maxRead { - return dn - } - } - return 0 -} - -var logger = log.New(ioutil.Discard, "bleve.index.firestorm ", 0) - -// SetLog sets the logger used for logging -// by default log messages are sent to ioutil.Discard -func SetLog(l *log.Logger) { - logger = l -} diff --git a/vendor/github.com/blevesearch/bleve/index/firestorm/version.go b/vendor/github.com/blevesearch/bleve/index/firestorm/version.go deleted file mode 100644 index bc5717b..0000000 --- a/vendor/github.com/blevesearch/bleve/index/firestorm/version.go +++ /dev/null @@ -1,108 +0,0 @@ -// Copyright (c) 2015 Couchbase, Inc. -// Licensed under the Apache License, Version 2.0 (the "License"); you may not use this file -// except in compliance with the License. You may obtain a copy of the License at -// http://www.apache.org/licenses/LICENSE-2.0 -// Unless required by applicable law or agreed to in writing, software distributed under the -// License is distributed on an "AS IS" BASIS, WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, -// either express or implied. See the License for the specific language governing permissions -// and limitations under the License. - -package firestorm - -import ( - "fmt" - - "github.com/golang/protobuf/proto" - - "github.com/blevesearch/bleve/index/store" -) - -const Version uint64 = 1 - -var IncompatibleVersion = fmt.Errorf("incompatible version, %d is supported", Version) - -var VersionKey = []byte{'v'} - -type VersionRow struct { - value VersionValue -} - -func NewVersionRow(version uint64) *VersionRow { - rv := VersionRow{} - rv.value.Version = proto.Uint64(version) - return &rv -} - -func NewVersionRowV(val []byte) (*VersionRow, error) { - rv := VersionRow{} - err := rv.value.Unmarshal(val) - if err != nil { - return nil, err - } - return &rv, nil -} - -func (vr *VersionRow) KeySize() int { - return 1 -} - -func (vr *VersionRow) KeyTo(buf []byte) (int, error) { - buf[0] = VersionKey[0] - return 1, nil -} - -func (vr *VersionRow) Key() []byte { - return VersionKey -} - -func (vr *VersionRow) ValueSize() int { - return vr.value.Size() -} - -func (vr *VersionRow) ValueTo(buf []byte) (int, error) { - return vr.value.MarshalTo(buf) -} - -func (vr *VersionRow) Value() []byte { - buf := make([]byte, vr.ValueSize()) - n, _ := vr.value.MarshalTo(buf) - return buf[:n] -} - -func (vr *VersionRow) Version() uint64 { - return vr.value.GetVersion() -} - -func (f *Firestorm) checkVersion(reader store.KVReader) (newIndex bool, err error) { - value, err := reader.Get(VersionKey) - if err != nil { - return - } - - if value == nil { - newIndex = true - return - } - - var vr *VersionRow - vr, err = NewVersionRowV(value) - if err != nil { - return - } - - // assert correct version - if vr.Version() != Version { - err = IncompatibleVersion - return - } - - return -} - -func (f *Firestorm) storeVersion(writer store.KVWriter) error { - vr := NewVersionRow(Version) - wb := writer.NewBatch() - wb.Set(vr.Key(), vr.Value()) - err := writer.ExecuteBatch(wb) - return err -} diff --git a/vendor/github.com/blevesearch/bleve/index/firestorm/warmup.go b/vendor/github.com/blevesearch/bleve/index/firestorm/warmup.go deleted file mode 100644 index 4e20575..0000000 --- a/vendor/github.com/blevesearch/bleve/index/firestorm/warmup.go +++ /dev/null @@ -1,129 +0,0 @@ -// Copyright (c) 2015 Couchbase, Inc. -// Licensed under the Apache License, Version 2.0 (the "License"); you may not use this file -// except in compliance with the License. You may obtain a copy of the License at -// http://www.apache.org/licenses/LICENSE-2.0 -// Unless required by applicable law or agreed to in writing, software distributed under the -// License is distributed on an "AS IS" BASIS, WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, -// either express or implied. See the License for the specific language governing permissions -// and limitations under the License. - -package firestorm - -import ( - "bytes" - "fmt" - "sort" - "sync/atomic" - - "github.com/blevesearch/bleve/index/store" -) - -const IDFieldName = "_id" - -func (f *Firestorm) bootstrap() (err error) { - - kvwriter, err := f.store.Writer() - if err != nil { - return - } - defer func() { - if cerr := kvwriter.Close(); err == nil && cerr != nil { - err = cerr - } - }() - - // record version - err = f.storeVersion(kvwriter) - if err != nil { - return - } - // define _id field - _, idFieldRow := f.fieldIndexOrNewRow(IDFieldName) - - wb := kvwriter.NewBatch() - wb.Set(idFieldRow.Key(), idFieldRow.Value()) - err = kvwriter.ExecuteBatch(wb) - if err != nil { - return - } - - return -} - -func (f *Firestorm) warmup(reader store.KVReader) error { - // load all the existing fields - err := f.loadFields(reader) - if err != nil { - return err - } - - // walk the term frequency info for _id - // this allows us to find deleted doc numbers - // and seed the doc count - idField, existed := f.fieldCache.FieldNamed(IDFieldName, false) - if !existed { - return fmt.Errorf("_id field missing, cannot proceed") - } - - tfkPrefix := TermFreqIteratorStart(idField, nil) - - var tfk TermFreqRow - var lastDocId []byte - lastDocNumbers := make(DocNumberList, 1) - err = visitPrefix(reader, tfkPrefix, func(key, val []byte) (bool, error) { - err := tfk.ParseKey(key) - if err != nil { - return false, err - } - docID := tfk.DocID() - docNum := tfk.DocNum() - - if docNum > f.highDocNumber { - f.highDocNumber = docNum - } - if docNum > f.compensator.maxRead { - f.compensator.maxRead = docNum - } - - // check for consecutive records - if bytes.Compare(docID, lastDocId) == 0 { - lastDocNumbers = append(lastDocNumbers, docNum) - } else { - // new doc id - atomic.AddUint64(&f.docCount, 1) - - // last docID had multiple doc numbers - if len(lastDocNumbers) > 1 { - f.addOldDocNumbers(lastDocNumbers, lastDocId) - - // reset size to 1 - lastDocNumbers = make(DocNumberList, 1) - } - lastDocNumbers = lastDocNumbers[:1] - lastDocNumbers[0] = docNum - lastDocId = make([]byte, len(docID)) - copy(lastDocId, docID) - } - return true, nil - }) - if err != nil { - return err - } - - // be sure to finish up check on final row - if len(lastDocNumbers) > 1 { - f.addOldDocNumbers(lastDocNumbers, lastDocId) - } - - return nil -} - -func (f *Firestorm) addOldDocNumbers(docNumberList DocNumberList, docID []byte) { - sort.Sort(docNumberList) - // high doc number is OK, rest are deleted - for _, dn := range docNumberList[1:] { - // f.deletedDocNumbers.Add(dn, docID) - f.compensator.deletedDocNumbers.Set(uint(dn)) - f.garbageCollector.Notify(dn, docID) - } -} diff --git a/vendor/github.com/blevesearch/bleve/index/index.go b/vendor/github.com/blevesearch/bleve/index/index.go index 1515f9a..958e8a8 100644 --- a/vendor/github.com/blevesearch/bleve/index/index.go +++ b/vendor/github.com/blevesearch/bleve/index/index.go @@ -10,9 +10,9 @@ package index import ( + "bytes" "encoding/json" "fmt" - "time" "github.com/blevesearch/bleve/document" "github.com/blevesearch/bleve/index/store" @@ -24,8 +24,6 @@ type Index interface { Open() error Close() error - DocCount() (uint64, error) - Update(doc *document.Document) error Delete(id string) error Batch(batch *Batch) error @@ -33,10 +31,6 @@ type Index interface { SetInternal(key, val []byte) error DeleteInternal(key []byte) error - DumpAll() chan interface{} - DumpDoc(id string) chan interface{} - DumpFields() chan interface{} - // Reader returns a low-level accessor on the index data. Close it to // release associated resources. Reader() (IndexReader, error) @@ -49,25 +43,14 @@ type Index interface { Advanced() (store.KVStore, error) } -// AsyncIndex is an interface for indexes which perform -// some important operations asynchronously. -type AsyncIndex interface { - // Wait will block until asynchronous operations started - // before this call have finished or until the specified - // timeout has been reached. If the timeout is reached - // an error is returned. - Wait(timeout time.Duration) error -} - type IndexReader interface { - TermFieldReader(term []byte, field string) (TermFieldReader, error) + TermFieldReader(term []byte, field string, includeFreq, includeNorm, includeTermVectors bool) (TermFieldReader, error) - // DocIDReader returns an iterator over documents which identifiers are - // greater than or equal to start and smaller than end. Set start to the - // empty string to iterate from the first document, end to the empty string - // to iterate to the last one. + // DocIDReader returns an iterator over all doc ids // The caller must close returned instance to release associated resources. - DocIDReader(start, end string) (DocIDReader, error) + DocIDReaderAll() (DocIDReader, error) + + DocIDReaderOnly(ids []string) (DocIDReader, error) FieldDict(field string) (FieldDict, error) @@ -76,19 +59,47 @@ type IndexReader interface { FieldDictPrefix(field string, termPrefix []byte) (FieldDict, error) Document(id string) (*document.Document, error) - DocumentFieldTerms(id string) (FieldTerms, error) + DocumentFieldTerms(id IndexInternalID, fields []string) (FieldTerms, error) Fields() ([]string, error) GetInternal(key []byte) ([]byte, error) - DocCount() uint64 + DocCount() (uint64, error) + + ExternalID(id IndexInternalID) (string, error) + InternalID(id string) (IndexInternalID, error) + + DumpAll() chan interface{} + DumpDoc(id string) chan interface{} + DumpFields() chan interface{} Close() error } +// FieldTerms contains the terms used by a document, keyed by field type FieldTerms map[string][]string +// FieldsNotYetCached returns a list of fields not yet cached out of a larger list of fields +func (f FieldTerms) FieldsNotYetCached(fields []string) []string { + var rv []string + for _, field := range fields { + if _, ok := f[field]; !ok { + rv = append(rv, field) + } + } + return rv +} + +// Merge will combine two FieldTerms +// it assumes that the terms lists are complete (thus do not need to be merged) +// field terms from the other list always replace the ones in the receiver +func (f FieldTerms) Merge(other FieldTerms) { + for field, terms := range other { + f[field] = terms + } +} + type TermFieldVector struct { Field string ArrayPositions []uint64 @@ -97,25 +108,48 @@ type TermFieldVector struct { End uint64 } +// IndexInternalID is an opaque document identifier interal to the index impl +type IndexInternalID []byte + +func (id IndexInternalID) Equals(other IndexInternalID) bool { + return id.Compare(other) == 0 +} + +func (id IndexInternalID) Compare(other IndexInternalID) int { + return bytes.Compare(id, other) +} + type TermFieldDoc struct { Term string - ID string + ID IndexInternalID Freq uint64 Norm float64 Vectors []*TermFieldVector } +// Reset allows an already allocated TermFieldDoc to be reused +func (tfd *TermFieldDoc) Reset() *TermFieldDoc { + // remember the []byte used for the ID + id := tfd.ID + // idiom to copy over from empty TermFieldDoc (0 allocations) + *tfd = TermFieldDoc{} + // reuse the []byte already allocated (and reset len to 0) + tfd.ID = id[:0] + return tfd +} + // TermFieldReader is the interface exposing the enumeration of documents // containing a given term in a given field. Documents are returned in byte // lexicographic order over their identifiers. type TermFieldReader interface { // Next returns the next document containing the term in this field, or nil - // when it reaches the end of the enumeration. - Next() (*TermFieldDoc, error) + // when it reaches the end of the enumeration. The preAlloced TermFieldDoc + // is optional, and when non-nil, will be used instead of allocating memory. + Next(preAlloced *TermFieldDoc) (*TermFieldDoc, error) // Advance resets the enumeration at specified document or its immediate // follower. - Advance(ID string) (*TermFieldDoc, error) + Advance(ID IndexInternalID, preAlloced *TermFieldDoc) (*TermFieldDoc, error) // Count returns the number of documents contains the term in this field. Count() uint64 @@ -135,15 +169,15 @@ type FieldDict interface { // DocIDReader is the interface exposing enumeration of documents identifiers. // Close the reader to release associated resources. type DocIDReader interface { - // Next returns the next document identifier in ascending lexicographic - // byte order, or io.EOF when the end of the sequence is reached. - Next() (string, error) + // Next returns the next document internal identifier in the natural + // index order, or io.EOF when the end of the sequence is reached. + Next() (IndexInternalID, error) - // Advance resets the iteration to the first identifier greater than or - // equal to ID. If ID is smaller than the start of the range, the iteration + // Advance resets the iteration to the first internal identifier greater than + // or equal to ID. If ID is smaller than the start of the range, the iteration // will start there instead. If ID is greater than or equal to the end of // the range, Next() call will return io.EOF. - Advance(ID string) (string, error) + Advance(ID IndexInternalID) (IndexInternalID, error) Close() error } diff --git a/vendor/github.com/blevesearch/bleve/index/store/boltdb/store.go b/vendor/github.com/blevesearch/bleve/index/store/boltdb/store.go index 26f320f..fa89383 100644 --- a/vendor/github.com/blevesearch/bleve/index/store/boltdb/store.go +++ b/vendor/github.com/blevesearch/bleve/index/store/boltdb/store.go @@ -18,22 +18,28 @@ package boltdb import ( + "bytes" "encoding/json" "fmt" + "os" "github.com/blevesearch/bleve/index/store" "github.com/blevesearch/bleve/registry" "github.com/boltdb/bolt" ) -const Name = "boltdb" +const ( + Name = "boltdb" + defaultCompactBatchSize = 100 +) type Store struct { - path string - bucket string - db *bolt.DB - noSync bool - mo store.MergeOperator + path string + bucket string + db *bolt.DB + noSync bool + fillPercent float64 + mo store.MergeOperator } func New(mo store.MergeOperator, config map[string]interface{}) (store.KVStore, error) { @@ -41,6 +47,9 @@ func New(mo store.MergeOperator, config map[string]interface{}) (store.KVStore, if !ok { return nil, fmt.Errorf("must specify path") } + if path == "" { + return nil, os.ErrInvalid + } bucket, ok := config["bucket"].(string) if !ok { @@ -49,27 +58,41 @@ func New(mo store.MergeOperator, config map[string]interface{}) (store.KVStore, noSync, _ := config["nosync"].(bool) - db, err := bolt.Open(path, 0600, nil) + fillPercent, ok := config["fillPercent"].(float64) + if !ok { + fillPercent = bolt.DefaultFillPercent + } + + bo := &bolt.Options{} + ro, ok := config["read_only"].(bool) + if ok { + bo.ReadOnly = ro + } + + db, err := bolt.Open(path, 0600, bo) if err != nil { return nil, err } db.NoSync = noSync - err = db.Update(func(tx *bolt.Tx) error { - _, err := tx.CreateBucketIfNotExists([]byte(bucket)) + if !bo.ReadOnly { + err = db.Update(func(tx *bolt.Tx) error { + _, err := tx.CreateBucketIfNotExists([]byte(bucket)) - return err - }) - if err != nil { - return nil, err + return err + }) + if err != nil { + return nil, err + } } rv := Store{ - path: path, - bucket: bucket, - db: db, - mo: mo, - noSync: noSync, + path: path, + bucket: bucket, + db: db, + mo: mo, + noSync: noSync, + fillPercent: fillPercent, } return &rv, nil } @@ -102,6 +125,46 @@ func (bs *Store) Stats() json.Marshaler { } } +// CompactWithBatchSize removes DictionaryTerm entries with a count of zero (in batchSize batches) +// Removing entries is a workaround for github issue #374. +func (bs *Store) CompactWithBatchSize(batchSize int) error { + for { + cnt := 0 + err := bs.db.Batch(func(tx *bolt.Tx) error { + c := tx.Bucket([]byte(bs.bucket)).Cursor() + prefix := []byte("d") + + for k, v := c.Seek(prefix); bytes.HasPrefix(k, prefix); k, v = c.Next() { + if bytes.Equal(v, []byte{0}) { + cnt++ + if err := c.Delete(); err != nil { + return err + } + if cnt == batchSize { + break + } + } + + } + return nil + }) + if err != nil { + return err + } + + if cnt == 0 { + break + } + } + return nil +} + +// Compact calls CompactWithBatchSize with a default batch size of 100. This is a workaround +// for github issue #374. +func (bs *Store) Compact() error { + return bs.CompactWithBatchSize(defaultCompactBatchSize) +} + func init() { registry.RegisterKVStore(Name, New) } diff --git a/vendor/github.com/blevesearch/bleve/index/store/boltdb/writer.go b/vendor/github.com/blevesearch/bleve/index/store/boltdb/writer.go index 4ba1d08..7c234c2 100644 --- a/vendor/github.com/blevesearch/bleve/index/store/boltdb/writer.go +++ b/vendor/github.com/blevesearch/bleve/index/store/boltdb/writer.go @@ -27,7 +27,7 @@ func (w *Writer) NewBatchEx(options store.KVBatchOptions) ([]byte, store.KVBatch return make([]byte, options.TotalBytes), w.NewBatch(), nil } -func (w *Writer) ExecuteBatch(batch store.KVBatch) error { +func (w *Writer) ExecuteBatch(batch store.KVBatch) (err error) { emulatedBatch, ok := batch.(*store.EmulatedBatch) if !ok { @@ -36,39 +36,53 @@ func (w *Writer) ExecuteBatch(batch store.KVBatch) error { tx, err := w.store.db.Begin(true) if err != nil { - return err + return } + // defer function to ensure that once started, + // we either Commit tx or Rollback + defer func() { + // if nothing went wrong, commit + if err == nil { + // careful to catch error here too + err = tx.Commit() + } else { + // caller should see error that caused abort, + // not success or failure of Rollback itself + _ = tx.Rollback() + } + }() bucket := tx.Bucket([]byte(w.store.bucket)) + bucket.FillPercent = w.store.fillPercent for k, mergeOps := range emulatedBatch.Merger.Merges { kb := []byte(k) existingVal := bucket.Get(kb) mergedVal, fullMergeOk := w.store.mo.FullMerge(kb, existingVal, mergeOps) if !fullMergeOk { - return fmt.Errorf("merge operator returned failure") + err = fmt.Errorf("merge operator returned failure") + return } err = bucket.Put(kb, mergedVal) if err != nil { - return err + return } } for _, op := range emulatedBatch.Ops { if op.V != nil { - err := bucket.Put(op.K, op.V) + err = bucket.Put(op.K, op.V) if err != nil { - return err + return } } else { - err := bucket.Delete(op.K) + err = bucket.Delete(op.K) if err != nil { - return err + return } } } - - return tx.Commit() + return } func (w *Writer) Close() error { diff --git a/vendor/github.com/blevesearch/bleve/index/store/gtreap/store.go b/vendor/github.com/blevesearch/bleve/index/store/gtreap/store.go index 7b0048f..0a1b825 100644 --- a/vendor/github.com/blevesearch/bleve/index/store/gtreap/store.go +++ b/vendor/github.com/blevesearch/bleve/index/store/gtreap/store.go @@ -17,6 +17,8 @@ package gtreap import ( "bytes" + "fmt" + "os" "sync" "github.com/blevesearch/bleve/index/store" @@ -42,6 +44,14 @@ func itemCompare(a, b interface{}) int { } func New(mo store.MergeOperator, config map[string]interface{}) (store.KVStore, error) { + path, ok := config["path"].(string) + if !ok { + return nil, fmt.Errorf("must specify path") + } + if path != "" { + return nil, os.ErrInvalid + } + rv := Store{ t: gtreap.NewTreap(itemCompare), mo: mo, diff --git a/vendor/github.com/blevesearch/bleve/index/upside_down/dump.go b/vendor/github.com/blevesearch/bleve/index/upside_down/dump.go index 023ae45..905269a 100644 --- a/vendor/github.com/blevesearch/bleve/index/upside_down/dump.go +++ b/vendor/github.com/blevesearch/bleve/index/upside_down/dump.go @@ -21,7 +21,7 @@ import ( // if your application relies on them, you're doing something wrong // they may change or be removed at any time -func (udc *UpsideDownCouch) dumpPrefix(kvreader store.KVReader, rv chan interface{}, prefix []byte) { +func dumpPrefix(kvreader store.KVReader, rv chan interface{}, prefix []byte) { start := prefix if start == nil { start = []byte{0} @@ -51,7 +51,7 @@ func (udc *UpsideDownCouch) dumpPrefix(kvreader store.KVReader, rv chan interfac } } -func (udc *UpsideDownCouch) dumpRange(kvreader store.KVReader, rv chan interface{}, start, end []byte) { +func dumpRange(kvreader store.KVReader, rv chan interface{}, start, end []byte) { it := kvreader.RangeIterator(start, end) defer func() { cerr := it.Close() @@ -77,48 +77,20 @@ func (udc *UpsideDownCouch) dumpRange(kvreader store.KVReader, rv chan interface } } -func (udc *UpsideDownCouch) DumpAll() chan interface{} { +func (i *IndexReader) DumpAll() chan interface{} { rv := make(chan interface{}) go func() { defer close(rv) - - // start an isolated reader for use during the dump - kvreader, err := udc.store.Reader() - if err != nil { - rv <- err - return - } - defer func() { - cerr := kvreader.Close() - if cerr != nil { - rv <- cerr - } - }() - - udc.dumpRange(kvreader, rv, nil, nil) + dumpRange(i.kvreader, rv, nil, nil) }() return rv } -func (udc *UpsideDownCouch) DumpFields() chan interface{} { +func (i *IndexReader) DumpFields() chan interface{} { rv := make(chan interface{}) go func() { defer close(rv) - - // start an isolated reader for use during the dump - kvreader, err := udc.store.Reader() - if err != nil { - rv <- err - return - } - defer func() { - cerr := kvreader.Close() - if cerr != nil { - rv <- cerr - } - }() - - udc.dumpPrefix(kvreader, rv, []byte{'f'}) + dumpPrefix(i.kvreader, rv, []byte{'f'}) }() return rv } @@ -130,7 +102,7 @@ func (k keyset) Swap(i, j int) { k[i], k[j] = k[j], k[i] } func (k keyset) Less(i, j int) bool { return bytes.Compare(k[i], k[j]) < 0 } // DumpDoc returns all rows in the index related to this doc id -func (udc *UpsideDownCouch) DumpDoc(id string) chan interface{} { +func (i *IndexReader) DumpDoc(id string) chan interface{} { idBytes := []byte(id) rv := make(chan interface{}) @@ -138,20 +110,7 @@ func (udc *UpsideDownCouch) DumpDoc(id string) chan interface{} { go func() { defer close(rv) - // start an isolated reader for use during the dump - kvreader, err := udc.store.Reader() - if err != nil { - rv <- err - return - } - defer func() { - cerr := kvreader.Close() - if cerr != nil { - rv <- cerr - } - }() - - back, err := udc.backIndexRowForDoc(kvreader, id) + back, err := backIndexRowForDoc(i.kvreader, []byte(id)) if err != nil { rv <- err return @@ -172,11 +131,11 @@ func (udc *UpsideDownCouch) DumpDoc(id string) chan interface{} { // first add all the stored rows storedRowPrefix := NewStoredRow(idBytes, 0, []uint64{}, 'x', []byte{}).ScanPrefixForDoc() - udc.dumpPrefix(kvreader, rv, storedRowPrefix) + dumpPrefix(i.kvreader, rv, storedRowPrefix) // now walk term keys in order and add them as well if len(keys) > 0 { - it := kvreader.RangeIterator(keys[0], nil) + it := i.kvreader.RangeIterator(keys[0], nil) defer func() { cerr := it.Close() if cerr != nil { diff --git a/vendor/github.com/blevesearch/bleve/index/upside_down/index_reader.go b/vendor/github.com/blevesearch/bleve/index/upside_down/index_reader.go index fb43a86..6cf638b 100644 --- a/vendor/github.com/blevesearch/bleve/index/upside_down/index_reader.go +++ b/vendor/github.com/blevesearch/bleve/index/upside_down/index_reader.go @@ -21,12 +21,12 @@ type IndexReader struct { docCount uint64 } -func (i *IndexReader) TermFieldReader(term []byte, fieldName string) (index.TermFieldReader, error) { +func (i *IndexReader) TermFieldReader(term []byte, fieldName string, includeFreq, includeNorm, includeTermVectors bool) (index.TermFieldReader, error) { fieldIndex, fieldExists := i.index.fieldCache.FieldNamed(fieldName, false) if fieldExists { - return newUpsideDownCouchTermFieldReader(i, term, uint16(fieldIndex)) + return newUpsideDownCouchTermFieldReader(i, term, uint16(fieldIndex), includeFreq, includeNorm, includeTermVectors) } - return newUpsideDownCouchTermFieldReader(i, []byte{ByteSeparator}, ^uint16(0)) + return newUpsideDownCouchTermFieldReader(i, []byte{ByteSeparator}, ^uint16(0), includeFreq, includeNorm, includeTermVectors) } func (i *IndexReader) FieldDict(fieldName string) (index.FieldDict, error) { @@ -45,14 +45,18 @@ func (i *IndexReader) FieldDictPrefix(fieldName string, termPrefix []byte) (inde return i.FieldDictRange(fieldName, termPrefix, termPrefix) } -func (i *IndexReader) DocIDReader(start, end string) (index.DocIDReader, error) { - return newUpsideDownCouchDocIDReader(i, start, end) +func (i *IndexReader) DocIDReaderAll() (index.DocIDReader, error) { + return newUpsideDownCouchDocIDReader(i) +} + +func (i *IndexReader) DocIDReaderOnly(ids []string) (index.DocIDReader, error) { + return newUpsideDownCouchDocIDReaderOnly(i, ids) } func (i *IndexReader) Document(id string) (doc *document.Document, err error) { // first hit the back index to confirm doc exists var backIndexRow *BackIndexRow - backIndexRow, err = i.index.backIndexRowForDoc(i.kvreader, id) + backIndexRow, err = backIndexRowForDoc(i.kvreader, []byte(id)) if err != nil { return } @@ -92,20 +96,31 @@ func (i *IndexReader) Document(id string) (doc *document.Document, err error) { return } -func (i *IndexReader) DocumentFieldTerms(id string) (index.FieldTerms, error) { - back, err := i.index.backIndexRowForDoc(i.kvreader, id) +func (i *IndexReader) DocumentFieldTerms(id index.IndexInternalID, fields []string) (index.FieldTerms, error) { + back, err := backIndexRowForDoc(i.kvreader, id) if err != nil { return nil, err } - rv := make(index.FieldTerms, len(back.termEntries)) - for _, entry := range back.termEntries { - fieldName := i.index.fieldCache.FieldIndexed(uint16(*entry.Field)) - terms, ok := rv[fieldName] - if !ok { - terms = make([]string, 0) + if back == nil { + return nil, nil + } + rv := make(index.FieldTerms, len(fields)) + fieldsMap := make(map[uint16]string, len(fields)) + for _, f := range fields { + id, ok := i.index.fieldCache.FieldNamed(f, false) + if ok { + fieldsMap[id] = f + } + } + for _, entry := range back.termEntries { + if field, ok := fieldsMap[uint16(*entry.Field)]; ok { + terms, ok := rv[field] + if !ok { + terms = make([]string, 0) + } + terms = append(terms, *entry.Term) + rv[field] = terms } - terms = append(terms, *entry.Term) - rv[fieldName] = terms } return rv, nil } @@ -144,14 +159,22 @@ func (i *IndexReader) GetInternal(key []byte) ([]byte, error) { return i.kvreader.Get(internalRow.Key()) } -func (i *IndexReader) DocCount() uint64 { - return i.docCount +func (i *IndexReader) DocCount() (uint64, error) { + return i.docCount, nil } func (i *IndexReader) Close() error { return i.kvreader.Close() } +func (i *IndexReader) ExternalID(id index.IndexInternalID) (string, error) { + return string(id), nil +} + +func (i *IndexReader) InternalID(id string) (index.IndexInternalID, error) { + return index.IndexInternalID(id), nil +} + func incrementBytes(in []byte) []byte { rv := make([]byte, len(in)) copy(rv, in) diff --git a/vendor/github.com/blevesearch/bleve/index/upside_down/reader.go b/vendor/github.com/blevesearch/bleve/index/upside_down/reader.go index 07de493..8129310 100644 --- a/vendor/github.com/blevesearch/bleve/index/upside_down/reader.go +++ b/vendor/github.com/blevesearch/bleve/index/upside_down/reader.go @@ -10,6 +10,8 @@ package upside_down import ( + "bytes" + "sort" "sync/atomic" "github.com/blevesearch/bleve/index" @@ -17,14 +19,16 @@ import ( ) type UpsideDownCouchTermFieldReader struct { + count uint64 indexReader *IndexReader iterator store.KVIterator - count uint64 term []byte + tfrNext *TermFrequencyRow + keyBuf []byte field uint16 } -func newUpsideDownCouchTermFieldReader(indexReader *IndexReader, term []byte, field uint16) (*UpsideDownCouchTermFieldReader, error) { +func newUpsideDownCouchTermFieldReader(indexReader *IndexReader, term []byte, field uint16, includeFreq, includeNorm, includeTermVectors bool) (*UpsideDownCouchTermFieldReader, error) { dictionaryRow := NewDictionaryRow(term, field, 0) val, err := indexReader.kvreader.Get(dictionaryRow.Key()) if err != nil { @@ -33,9 +37,10 @@ func newUpsideDownCouchTermFieldReader(indexReader *IndexReader, term []byte, fi if val == nil { atomic.AddUint64(&indexReader.index.stats.termSearchersStarted, uint64(1)) return &UpsideDownCouchTermFieldReader{ - count: 0, - term: term, - field: field, + count: 0, + term: term, + tfrNext: &TermFrequencyRow{}, + field: field, }, nil } @@ -61,45 +66,75 @@ func (r *UpsideDownCouchTermFieldReader) Count() uint64 { return r.count } -func (r *UpsideDownCouchTermFieldReader) Next() (*index.TermFieldDoc, error) { +func (r *UpsideDownCouchTermFieldReader) Next(preAlloced *index.TermFieldDoc) (*index.TermFieldDoc, error) { if r.iterator != nil { + // We treat tfrNext also like an initialization flag, which + // tells us whether we need to invoke the underlying + // iterator.Next(). The first time, don't call iterator.Next(). + if r.tfrNext != nil { + r.iterator.Next() + } else { + r.tfrNext = &TermFrequencyRow{} + } key, val, valid := r.iterator.Current() if valid { - tfr, err := NewTermFrequencyRowKV(key, val) + tfr := r.tfrNext + err := tfr.parseKDoc(key, r.term) if err != nil { return nil, err } - rv := index.TermFieldDoc{ - ID: string(tfr.doc), - Freq: tfr.freq, - Norm: float64(tfr.norm), - Vectors: r.indexReader.index.termFieldVectorsFromTermVectors(tfr.vectors), + err = tfr.parseV(val) + if err != nil { + return nil, err } - r.iterator.Next() - return &rv, nil + rv := preAlloced + if rv == nil { + rv = &index.TermFieldDoc{} + } + rv.ID = append(rv.ID, tfr.doc...) + rv.Freq = tfr.freq + rv.Norm = float64(tfr.norm) + if tfr.vectors != nil { + rv.Vectors = r.indexReader.index.termFieldVectorsFromTermVectors(tfr.vectors) + } + return rv, nil } } return nil, nil } -func (r *UpsideDownCouchTermFieldReader) Advance(docID string) (*index.TermFieldDoc, error) { +func (r *UpsideDownCouchTermFieldReader) Advance(docID index.IndexInternalID, preAlloced *index.TermFieldDoc) (rv *index.TermFieldDoc, err error) { if r.iterator != nil { - tfr := NewTermFrequencyRow(r.term, r.field, []byte(docID), 0, 0) - r.iterator.Seek(tfr.Key()) + if r.tfrNext == nil { + r.tfrNext = &TermFrequencyRow{} + } + tfr := InitTermFrequencyRow(r.tfrNext, r.term, r.field, docID, 0, 0) + r.keyBuf, err = tfr.KeyAppendTo(r.keyBuf[:0]) + if err != nil { + return nil, err + } + r.iterator.Seek(r.keyBuf) key, val, valid := r.iterator.Current() if valid { - tfr, err := NewTermFrequencyRowKV(key, val) + err := tfr.parseKDoc(key, r.term) if err != nil { return nil, err } - rv := index.TermFieldDoc{ - ID: string(tfr.doc), - Freq: tfr.freq, - Norm: float64(tfr.norm), - Vectors: r.indexReader.index.termFieldVectorsFromTermVectors(tfr.vectors), + err = tfr.parseV(val) + if err != nil { + return nil, err } - r.iterator.Next() - return &rv, nil + rv = preAlloced + if rv == nil { + rv = &index.TermFieldDoc{} + } + rv.ID = append(rv.ID, tfr.doc...) + rv.Freq = tfr.freq + rv.Norm = float64(tfr.norm) + if tfr.vectors != nil { + rv.Vectors = r.indexReader.index.termFieldVectorsFromTermVectors(tfr.vectors) + } + return rv, nil } } return nil, nil @@ -115,17 +150,16 @@ func (r *UpsideDownCouchTermFieldReader) Close() error { type UpsideDownCouchDocIDReader struct { indexReader *IndexReader iterator store.KVIterator + only []string + onlyPos int + onlyMode bool } -func newUpsideDownCouchDocIDReader(indexReader *IndexReader, start, end string) (*UpsideDownCouchDocIDReader, error) { - startBytes := []byte(start) - if start == "" { - startBytes = []byte{0x0} - } - endBytes := []byte(end) - if end == "" { - endBytes = []byte{0xff} - } +func newUpsideDownCouchDocIDReader(indexReader *IndexReader) (*UpsideDownCouchDocIDReader, error) { + + startBytes := []byte{0x0} + endBytes := []byte{0xff} + bisr := NewBackIndexRow(startBytes, nil, nil) bier := NewBackIndexRow(endBytes, nil, nil) it := indexReader.kvreader.RangeIterator(bisr.Key(), bier.Key()) @@ -136,37 +170,149 @@ func newUpsideDownCouchDocIDReader(indexReader *IndexReader, start, end string) }, nil } -func (r *UpsideDownCouchDocIDReader) Next() (string, error) { - key, val, valid := r.iterator.Current() - if valid { - br, err := NewBackIndexRowKV(key, val) - if err != nil { - return "", err - } - rv := string(br.doc) - r.iterator.Next() - return rv, nil +func newUpsideDownCouchDocIDReaderOnly(indexReader *IndexReader, ids []string) (*UpsideDownCouchDocIDReader, error) { + // ensure ids are sorted + sort.Strings(ids) + startBytes := []byte{0x0} + if len(ids) > 0 { + startBytes = []byte(ids[0]) } - return "", nil + endBytes := []byte{0xff} + if len(ids) > 0 { + endBytes = incrementBytes([]byte(ids[len(ids)-1])) + } + bisr := NewBackIndexRow(startBytes, nil, nil) + bier := NewBackIndexRow(endBytes, nil, nil) + it := indexReader.kvreader.RangeIterator(bisr.Key(), bier.Key()) + + return &UpsideDownCouchDocIDReader{ + indexReader: indexReader, + iterator: it, + only: ids, + onlyMode: true, + }, nil } -func (r *UpsideDownCouchDocIDReader) Advance(docID string) (string, error) { - bir := NewBackIndexRow([]byte(docID), nil, nil) - r.iterator.Seek(bir.Key()) +func (r *UpsideDownCouchDocIDReader) Next() (index.IndexInternalID, error) { key, val, valid := r.iterator.Current() - if valid { - br, err := NewBackIndexRowKV(key, val) - if err != nil { - return "", err + + if r.onlyMode { + var rv index.IndexInternalID + for valid && r.onlyPos < len(r.only) { + br, err := NewBackIndexRowKV(key, val) + if err != nil { + return nil, err + } + if !bytes.Equal(br.doc, []byte(r.only[r.onlyPos])) { + ok := r.nextOnly() + if !ok { + return nil, nil + } + r.iterator.Seek(NewBackIndexRow([]byte(r.only[r.onlyPos]), nil, nil).Key()) + key, val, valid = r.iterator.Current() + continue + } else { + rv = append([]byte(nil), br.doc...) + break + } + } + if valid && r.onlyPos < len(r.only) { + ok := r.nextOnly() + if ok { + r.iterator.Seek(NewBackIndexRow([]byte(r.only[r.onlyPos]), nil, nil).Key()) + } + return rv, nil + } + + } else { + if valid { + br, err := NewBackIndexRowKV(key, val) + if err != nil { + return nil, err + } + rv := append([]byte(nil), br.doc...) + r.iterator.Next() + return rv, nil } - rv := string(br.doc) - r.iterator.Next() - return rv, nil } - return "", nil + return nil, nil +} + +func (r *UpsideDownCouchDocIDReader) Advance(docID index.IndexInternalID) (index.IndexInternalID, error) { + + if r.onlyMode { + r.onlyPos = sort.SearchStrings(r.only, string(docID)) + if r.onlyPos >= len(r.only) { + // advanced to key after our last only key + return nil, nil + } + r.iterator.Seek(NewBackIndexRow([]byte(r.only[r.onlyPos]), nil, nil).Key()) + key, val, valid := r.iterator.Current() + + var rv index.IndexInternalID + for valid && r.onlyPos < len(r.only) { + br, err := NewBackIndexRowKV(key, val) + if err != nil { + return nil, err + } + if !bytes.Equal(br.doc, []byte(r.only[r.onlyPos])) { + // the only key we seek'd to didn't exist + // now look for the closest key that did exist in only + r.onlyPos = sort.SearchStrings(r.only, string(br.doc)) + if r.onlyPos >= len(r.only) { + // advanced to key after our last only key + return nil, nil + } + // now seek to this new only key + r.iterator.Seek(NewBackIndexRow([]byte(r.only[r.onlyPos]), nil, nil).Key()) + key, val, valid = r.iterator.Current() + continue + } else { + rv = append([]byte(nil), br.doc...) + break + } + } + if valid && r.onlyPos < len(r.only) { + ok := r.nextOnly() + if ok { + r.iterator.Seek(NewBackIndexRow([]byte(r.only[r.onlyPos]), nil, nil).Key()) + } + return rv, nil + } + } else { + bir := NewBackIndexRow(docID, nil, nil) + r.iterator.Seek(bir.Key()) + key, val, valid := r.iterator.Current() + if valid { + br, err := NewBackIndexRowKV(key, val) + if err != nil { + return nil, err + } + rv := append([]byte(nil), br.doc...) + r.iterator.Next() + return rv, nil + } + } + return nil, nil } func (r *UpsideDownCouchDocIDReader) Close() error { atomic.AddUint64(&r.indexReader.index.stats.termSearchersFinished, uint64(1)) return r.iterator.Close() } + +// move the r.only pos forward one, skipping duplicates +// return true if there is more data, or false if we got to the end of the list +func (r *UpsideDownCouchDocIDReader) nextOnly() bool { + + // advance 1 position, until we see a different key + // it's already sorted, so this skips duplicates + start := r.onlyPos + r.onlyPos++ + for r.onlyPos < len(r.only) && r.only[r.onlyPos] == r.only[start] { + start = r.onlyPos + r.onlyPos++ + } + // inidicate if we got to the end of the list + return r.onlyPos < len(r.only) +} diff --git a/vendor/github.com/blevesearch/bleve/index/upside_down/row.go b/vendor/github.com/blevesearch/bleve/index/upside_down/row.go index 5685e52..667b6d4 100644 --- a/vendor/github.com/blevesearch/bleve/index/upside_down/row.go +++ b/vendor/github.com/blevesearch/bleve/index/upside_down/row.go @@ -350,11 +350,11 @@ func (tv *TermVector) String() string { type TermFrequencyRow struct { term []byte - field uint16 doc []byte freq uint64 - norm float32 vectors []*TermVector + norm float32 + field uint16 } func (tfr *TermFrequencyRow) Term() []byte { @@ -408,6 +408,15 @@ func (tfr *TermFrequencyRow) KeyTo(buf []byte) (int, error) { return 3 + termLen + 1 + docLen, nil } +func (tfr *TermFrequencyRow) KeyAppendTo(buf []byte) ([]byte, error) { + keySize := tfr.KeySize() + if cap(buf) < keySize { + buf = make([]byte, keySize) + } + actualSize, err := tfr.KeyTo(buf[0:keySize]) + return buf[0:actualSize], err +} + func (tfr *TermFrequencyRow) DictionaryRowKey() []byte { dr := NewDictionaryRow(tfr.term, tfr.field, 0) return dr.Key() @@ -461,6 +470,15 @@ func (tfr *TermFrequencyRow) String() string { return fmt.Sprintf("Term: `%s` Field: %d DocId: `%s` Frequency: %d Norm: %f Vectors: %v", string(tfr.term), tfr.field, string(tfr.doc), tfr.freq, tfr.norm, tfr.vectors) } +func InitTermFrequencyRow(tfr *TermFrequencyRow, term []byte, field uint16, docID []byte, freq uint64, norm float32) *TermFrequencyRow { + tfr.term = term + tfr.field = field + tfr.doc = docID + tfr.freq = freq + tfr.norm = norm + return tfr +} + func NewTermFrequencyRow(term []byte, field uint16, docID []byte, freq uint64, norm float32) *TermFrequencyRow { return &TermFrequencyRow{ term: term, @@ -483,36 +501,52 @@ func NewTermFrequencyRowWithTermVectors(term []byte, field uint16, docID []byte, } func NewTermFrequencyRowK(key []byte) (*TermFrequencyRow, error) { - rv := TermFrequencyRow{} + rv := &TermFrequencyRow{} + err := rv.parseK(key) + if err != nil { + return nil, err + } + return rv, nil +} + +func (tfr *TermFrequencyRow) parseK(key []byte) error { keyLen := len(key) if keyLen < 3 { - return nil, fmt.Errorf("invalid term frequency key, no valid field") + return fmt.Errorf("invalid term frequency key, no valid field") } - rv.field = binary.LittleEndian.Uint16(key[1:3]) + tfr.field = binary.LittleEndian.Uint16(key[1:3]) termEndPos := bytes.IndexByte(key[3:], ByteSeparator) if termEndPos < 0 { - return nil, fmt.Errorf("invalid term frequency key, no byte separator terminating term") + return fmt.Errorf("invalid term frequency key, no byte separator terminating term") } - rv.term = key[3 : 3+termEndPos] + tfr.term = key[3 : 3+termEndPos] - docLen := len(key) - (3 + termEndPos + 1) + docLen := keyLen - (3 + termEndPos + 1) if docLen < 1 { - return nil, fmt.Errorf("invalid term frequency key, empty docid") + return fmt.Errorf("invalid term frequency key, empty docid") } - rv.doc = key[3+termEndPos+1:] + tfr.doc = key[3+termEndPos+1:] - return &rv, nil + return nil +} + +func (tfr *TermFrequencyRow) parseKDoc(key []byte, term []byte) error { + tfr.doc = key[3+len(term)+1:] + if len(tfr.doc) <= 0 { + return fmt.Errorf("invalid term frequency key, empty docid") + } + + return nil } func (tfr *TermFrequencyRow) parseV(value []byte) error { - currOffset := 0 - bytesRead := 0 - tfr.freq, bytesRead = binary.Uvarint(value[currOffset:]) + var bytesRead int + tfr.freq, bytesRead = binary.Uvarint(value) if bytesRead <= 0 { return fmt.Errorf("invalid term frequency value, invalid frequency") } - currOffset += bytesRead + currOffset := bytesRead var norm uint64 norm, bytesRead = binary.Uvarint(value[currOffset:]) @@ -523,6 +557,7 @@ func (tfr *TermFrequencyRow) parseV(value []byte) error { tfr.norm = math.Float32frombits(uint32(norm)) + tfr.vectors = nil var field uint64 field, bytesRead = binary.Uvarint(value[currOffset:]) for bytesRead > 0 { diff --git a/vendor/github.com/blevesearch/bleve/index/upside_down/upside_down.go b/vendor/github.com/blevesearch/bleve/index/upside_down/upside_down.go index 5e9715a..ff73929 100644 --- a/vendor/github.com/blevesearch/bleve/index/upside_down/upside_down.go +++ b/vendor/github.com/blevesearch/bleve/index/upside_down/upside_down.go @@ -285,12 +285,6 @@ func (udc *UpsideDownCouch) batchRows(writer store.KVWriter, addRowsAll [][]Upsi return writer.ExecuteBatch(wb) } -func (udc *UpsideDownCouch) DocCount() (uint64, error) { - udc.m.RLock() - defer udc.m.RUnlock() - return udc.docCount, nil -} - func (udc *UpsideDownCouch) Open() (err error) { //acquire the write mutex for the duratin of Open() udc.writeMutex.Lock() @@ -439,7 +433,7 @@ func (udc *UpsideDownCouch) Update(doc *document.Document) (err error) { // first we lookup the backindex row for the doc id if it exists // lookup the back index row var backIndexRow *BackIndexRow - backIndexRow, err = udc.backIndexRowForDoc(kvreader, doc.ID) + backIndexRow, err = backIndexRowForDoc(kvreader, index.IndexInternalID(doc.ID)) if err != nil { _ = kvreader.Close() atomic.AddUint64(&udc.stats.errors, 1) @@ -627,7 +621,7 @@ func (udc *UpsideDownCouch) Delete(id string) (err error) { // first we lookup the backindex row for the doc id if it exists // lookup the back index row var backIndexRow *BackIndexRow - backIndexRow, err = udc.backIndexRowForDoc(kvreader, id) + backIndexRow, err = backIndexRowForDoc(kvreader, index.IndexInternalID(id)) if err != nil { _ = kvreader.Close() atomic.AddUint64(&udc.stats.errors, 1) @@ -695,36 +689,6 @@ func (udc *UpsideDownCouch) deleteSingle(id string, backIndexRow *BackIndexRow, return deleteRows } -func (udc *UpsideDownCouch) backIndexRowForDoc(kvreader store.KVReader, docID string) (*BackIndexRow, error) { - // use a temporary row structure to build key - tempRow := &BackIndexRow{ - doc: []byte(docID), - } - - keyBuf := GetRowBuffer() - if tempRow.KeySize() > len(keyBuf) { - keyBuf = make([]byte, 2*tempRow.KeySize()) - } - defer PutRowBuffer(keyBuf) - keySize, err := tempRow.KeyTo(keyBuf) - if err != nil { - return nil, err - } - - value, err := kvreader.Get(keyBuf[:keySize]) - if err != nil { - return nil, err - } - if value == nil { - return nil, nil - } - backIndexRow, err := NewBackIndexRowKV(keyBuf[:keySize], value) - if err != nil { - return nil, err - } - return backIndexRow, nil -} - func decodeFieldType(typ byte, name string, pos []uint64, value []byte) document.Field { switch typ { case 't': @@ -770,6 +734,10 @@ func (udc *UpsideDownCouch) termVectorsFromTokenFreq(field uint16, tf *analysis. } func (udc *UpsideDownCouch) termFieldVectorsFromTermVectors(in []*TermVector) []*index.TermFieldVector { + if len(in) <= 0 { + return nil + } + rv := make([]*index.TermFieldVector, len(in)) for i, tv := range in { @@ -829,7 +797,7 @@ func (udc *UpsideDownCouch) Batch(batch *index.Batch) (err error) { } for docID, doc := range batch.IndexOps { - backIndexRow, err := udc.backIndexRowForDoc(kvreader, docID) + backIndexRow, err := backIndexRowForDoc(kvreader, index.IndexInternalID(docID)) if err != nil { docBackIndexRowErr = err return @@ -1030,3 +998,33 @@ func (udc *UpsideDownCouch) fieldIndexOrNewRow(name string) (uint16, *FieldRow) func init() { registry.RegisterIndexType(Name, NewUpsideDownCouch) } + +func backIndexRowForDoc(kvreader store.KVReader, docID index.IndexInternalID) (*BackIndexRow, error) { + // use a temporary row structure to build key + tempRow := &BackIndexRow{ + doc: docID, + } + + keyBuf := GetRowBuffer() + if tempRow.KeySize() > len(keyBuf) { + keyBuf = make([]byte, 2*tempRow.KeySize()) + } + defer PutRowBuffer(keyBuf) + keySize, err := tempRow.KeyTo(keyBuf) + if err != nil { + return nil, err + } + + value, err := kvreader.Get(keyBuf[:keySize]) + if err != nil { + return nil, err + } + if value == nil { + return nil, nil + } + backIndexRow, err := NewBackIndexRowKV(keyBuf[:keySize], value) + if err != nil { + return nil, err + } + return backIndexRow, nil +} diff --git a/vendor/github.com/blevesearch/bleve/index_alias_impl.go b/vendor/github.com/blevesearch/bleve/index_alias_impl.go index 03f30f5..ea3d6fb 100644 --- a/vendor/github.com/blevesearch/bleve/index_alias_impl.go +++ b/vendor/github.com/blevesearch/bleve/index_alias_impl.go @@ -251,54 +251,6 @@ func (i *indexAliasImpl) FieldDictPrefix(field string, termPrefix []byte) (index }, nil } -func (i *indexAliasImpl) DumpAll() chan interface{} { - i.mutex.RLock() - defer i.mutex.RUnlock() - - if !i.open { - return nil - } - - err := i.isAliasToSingleIndex() - if err != nil { - return nil - } - - return i.indexes[0].DumpAll() -} - -func (i *indexAliasImpl) DumpDoc(id string) chan interface{} { - i.mutex.RLock() - defer i.mutex.RUnlock() - - if !i.open { - return nil - } - - err := i.isAliasToSingleIndex() - if err != nil { - return nil - } - - return i.indexes[0].DumpDoc(id) -} - -func (i *indexAliasImpl) DumpFields() chan interface{} { - i.mutex.RLock() - defer i.mutex.RUnlock() - - if !i.open { - return nil - } - - err := i.isAliasToSingleIndex() - if err != nil { - return nil - } - - return i.indexes[0].DumpFields() -} - func (i *indexAliasImpl) Close() error { i.mutex.Lock() defer i.mutex.Unlock() @@ -474,6 +426,7 @@ func createChildSearchRequest(req *SearchRequest) *SearchRequest { Fields: req.Fields, Facets: req.Facets, Explain: req.Explain, + Sort: req.Sort, } return &rv } @@ -568,8 +521,11 @@ func MultiSearch(ctx context.Context, req *SearchRequest, indexes ...Index) (*Se } } - // first sort it by score - sort.Sort(sr.Hits) + // sort all hits with the requested order + if len(req.Sort) > 0 { + sorter := newMultiSearchHitSorter(req.Sort, sr.Hits) + sort.Sort(sorter) + } // now skip over the correct From if req.From > 0 && len(sr.Hits) > req.From { @@ -645,3 +601,26 @@ func (f *indexAliasImplFieldDict) Close() error { defer f.index.mutex.RUnlock() return f.fieldDict.Close() } + +type multiSearchHitSorter struct { + hits search.DocumentMatchCollection + sort search.SortOrder + cachedScoring []bool + cachedDesc []bool +} + +func newMultiSearchHitSorter(sort search.SortOrder, hits search.DocumentMatchCollection) *multiSearchHitSorter { + return &multiSearchHitSorter{ + sort: sort, + hits: hits, + cachedScoring: sort.CacheIsScore(), + cachedDesc: sort.CacheDescending(), + } +} + +func (m *multiSearchHitSorter) Len() int { return len(m.hits) } +func (m *multiSearchHitSorter) Swap(i, j int) { m.hits[i], m.hits[j] = m.hits[j], m.hits[i] } +func (m *multiSearchHitSorter) Less(i, j int) bool { + c := m.sort.Compare(m.cachedScoring, m.cachedDesc, m.hits[i], m.hits[j]) + return c < 0 +} diff --git a/vendor/github.com/blevesearch/bleve/index_impl.go b/vendor/github.com/blevesearch/bleve/index_impl.go index dac4b62..aae26a0 100644 --- a/vendor/github.com/blevesearch/bleve/index_impl.go +++ b/vendor/github.com/blevesearch/bleve/index_impl.go @@ -22,12 +22,12 @@ import ( "github.com/blevesearch/bleve/document" "github.com/blevesearch/bleve/index" "github.com/blevesearch/bleve/index/store" - "github.com/blevesearch/bleve/index/store/gtreap" "github.com/blevesearch/bleve/index/upside_down" "github.com/blevesearch/bleve/registry" "github.com/blevesearch/bleve/search" "github.com/blevesearch/bleve/search/collectors" "github.com/blevesearch/bleve/search/facets" + "github.com/blevesearch/bleve/search/highlight" ) type indexImpl struct { @@ -49,49 +49,6 @@ func indexStorePath(path string) string { return path + string(os.PathSeparator) + storePath } -func newMemIndex(indexType string, mapping *IndexMapping) (*indexImpl, error) { - rv := indexImpl{ - path: "", - name: "mem", - m: mapping, - meta: newIndexMeta(indexType, gtreap.Name, nil), - stats: &IndexStat{}, - } - - // open the index - indexTypeConstructor := registry.IndexTypeConstructorByName(rv.meta.IndexType) - if indexTypeConstructor == nil { - return nil, ErrorUnknownIndexType - } - - var err error - rv.i, err = indexTypeConstructor(rv.meta.Storage, nil, Config.analysisQueue) - if err != nil { - return nil, err - } - err = rv.i.Open() - if err != nil { - return nil, err - } - - // now persist the mapping - mappingBytes, err := json.Marshal(mapping) - if err != nil { - return nil, err - } - err = rv.i.SetInternal(mappingInternalKey, mappingBytes) - if err != nil { - return nil, err - } - - // mark the index as open - rv.mutex.Lock() - defer rv.mutex.Unlock() - rv.open = true - indexStats.Register(&rv) - return &rv, nil -} - func newIndexUsing(path string, mapping *IndexMapping, indexType string, kvstore string, kvconfig map[string]interface{}) (*indexImpl, error) { // first validate the mapping err := mapping.Validate() @@ -99,14 +56,14 @@ func newIndexUsing(path string, mapping *IndexMapping, indexType string, kvstore return nil, err } - if path == "" { - return newMemIndex(indexType, mapping) - } - if kvconfig == nil { kvconfig = map[string]interface{}{} } + if kvstore == "" { + return nil, fmt.Errorf("bleve not configured for file based indexing") + } + rv := indexImpl{ path: path, name: path, @@ -115,13 +72,17 @@ func newIndexUsing(path string, mapping *IndexMapping, indexType string, kvstore } rv.stats = &IndexStat{i: &rv} // at this point there is hope that we can be successful, so save index meta - err = rv.meta.Save(path) - if err != nil { - return nil, err + if path != "" { + err = rv.meta.Save(path) + if err != nil { + return nil, err + } + kvconfig["create_if_missing"] = true + kvconfig["error_if_exists"] = true + kvconfig["path"] = indexStorePath(path) + } else { + kvconfig["path"] = "" } - kvconfig["create_if_missing"] = true - kvconfig["error_if_exists"] = true - kvconfig["path"] = indexStorePath(path) // open the index indexTypeConstructor := registry.IndexTypeConstructorByName(rv.meta.IndexType) @@ -349,7 +310,7 @@ func (i *indexImpl) Document(id string) (doc *document.Document, err error) { // DocCount returns the number of documents in the // index. -func (i *indexImpl) DocCount() (uint64, error) { +func (i *indexImpl) DocCount() (count uint64, err error) { i.mutex.RLock() defer i.mutex.RUnlock() @@ -357,7 +318,19 @@ func (i *indexImpl) DocCount() (uint64, error) { return 0, ErrorIndexClosed } - return i.i.DocCount() + // open a reader for this search + indexReader, err := i.i.Reader() + if err != nil { + return 0, fmt.Errorf("error opening index reader %v", err) + } + defer func() { + if cerr := indexReader.Close(); err == nil && cerr != nil { + err = cerr + } + }() + + count, err = indexReader.DocCount() + return } // Search executes a search request operation. @@ -378,7 +351,7 @@ func (i *indexImpl) SearchInContext(ctx context.Context, req *SearchRequest) (sr return nil, ErrorIndexClosed } - collector := collectors.NewTopScorerSkipCollector(req.Size, req.From) + collector := collectors.NewTopNCollector(req.Size, req.From, req.Sort) // open a reader for this search indexReader, err := i.i.Reader() @@ -429,16 +402,18 @@ func (i *indexImpl) SearchInContext(ctx context.Context, req *SearchRequest) (sr collector.SetFacetsBuilder(facetsBuilder) } - err = collector.Collect(ctx, searcher) + err = collector.Collect(ctx, searcher, indexReader) if err != nil { return nil, err } hits := collector.Results() + var highlighter highlight.Highlighter + if req.Highlight != nil { // get the right highlighter - highlighter, err := Config.Cache.HighlighterNamed(Config.DefaultHighlighter) + highlighter, err = Config.Cache.HighlighterNamed(Config.DefaultHighlighter) if err != nil { return nil, err } @@ -451,74 +426,62 @@ func (i *indexImpl) SearchInContext(ctx context.Context, req *SearchRequest) (sr if highlighter == nil { return nil, fmt.Errorf("no highlighter named `%s` registered", *req.Highlight.Style) } - - for _, hit := range hits { - doc, err := indexReader.Document(hit.ID) - if err == nil && doc != nil { - highlightFields := req.Highlight.Fields - if highlightFields == nil { - // add all fields with matches - highlightFields = make([]string, 0, len(hit.Locations)) - for k := range hit.Locations { - highlightFields = append(highlightFields, k) - } - } - - for _, hf := range highlightFields { - highlighter.BestFragmentsInField(hit, doc, hf, 1) - } - } else if err == nil { - // unexpected case, a doc ID that was found as a search hit - // was unable to be found during document lookup - return nil, ErrorIndexReadInconsistency - } - } } - if len(req.Fields) > 0 { - for _, hit := range hits { - // FIXME avoid loading doc second time - // if we already loaded it for highlighting + for _, hit := range hits { + if len(req.Fields) > 0 || highlighter != nil { doc, err := indexReader.Document(hit.ID) if err == nil && doc != nil { - for _, f := range req.Fields { - for _, docF := range doc.Fields { - if f == "*" || docF.Name() == f { - var value interface{} - switch docF := docF.(type) { - case *document.TextField: - value = string(docF.Value()) - case *document.NumericField: - num, err := docF.Number() - if err == nil { - value = num + if len(req.Fields) > 0 { + for _, f := range req.Fields { + for _, docF := range doc.Fields { + if f == "*" || docF.Name() == f { + var value interface{} + switch docF := docF.(type) { + case *document.TextField: + value = string(docF.Value()) + case *document.NumericField: + num, err := docF.Number() + if err == nil { + value = num + } + case *document.DateTimeField: + datetime, err := docF.DateTime() + if err == nil { + value = datetime.Format(time.RFC3339) + } + case *document.BooleanField: + boolean, err := docF.Boolean() + if err == nil { + value = boolean + } } - case *document.DateTimeField: - datetime, err := docF.DateTime() - if err == nil { - value = datetime.Format(time.RFC3339) + if value != nil { + hit.AddFieldValue(docF.Name(), value) } - case *document.BooleanField: - boolean, err := docF.Boolean() - if err == nil { - value = boolean - } - } - if value != nil { - hit.AddFieldValue(docF.Name(), value) } } } } + if highlighter != nil { + highlightFields := req.Highlight.Fields + if highlightFields == nil { + // add all fields with matches + highlightFields = make([]string, 0, len(hit.Locations)) + for k := range hit.Locations { + highlightFields = append(highlightFields, k) + } + } + for _, hf := range highlightFields { + highlighter.BestFragmentsInField(hit, doc, hf, 1) + } + } } else if doc == nil { // unexpected case, a doc ID that was found as a search hit // was unable to be found during document lookup return nil, ErrorIndexReadInconsistency } } - } - - for _, hit := range hits { if i.name != "" { hit.Index = i.name } @@ -656,52 +619,12 @@ func (i *indexImpl) FieldDictPrefix(field string, termPrefix []byte) (index.Fiel }, nil } -// DumpAll writes all index rows to a channel. -// INTERNAL: do not rely on this function, it is -// only intended to be used by the debug utilities -func (i *indexImpl) DumpAll() chan interface{} { - i.mutex.RLock() - defer i.mutex.RUnlock() - - if !i.open { - return nil - } - - return i.i.DumpAll() -} - -// DumpFields writes all field rows in the index -// to a channel. -// INTERNAL: do not rely on this function, it is -// only intended to be used by the debug utilities -func (i *indexImpl) DumpFields() chan interface{} { - i.mutex.RLock() - defer i.mutex.RUnlock() - - if !i.open { - return nil - } - return i.i.DumpFields() -} - -// DumpDoc writes all rows in the index associated -// with the specified identifier to a channel. -// INTERNAL: do not rely on this function, it is -// only intended to be used by the debug utilities -func (i *indexImpl) DumpDoc(id string) chan interface{} { - i.mutex.RLock() - defer i.mutex.RUnlock() - - if !i.open { - return nil - } - return i.i.DumpDoc(id) -} - func (i *indexImpl) Close() error { i.mutex.Lock() defer i.mutex.Unlock() + indexStats.UnRegister(i) + i.open = false return i.i.Close() } diff --git a/vendor/github.com/blevesearch/bleve/mapping_document.go b/vendor/github.com/blevesearch/bleve/mapping_document.go index 59a999f..5bea75a 100644 --- a/vendor/github.com/blevesearch/bleve/mapping_document.go +++ b/vendor/github.com/blevesearch/bleve/mapping_document.go @@ -325,6 +325,10 @@ func (dm *DocumentMapping) walkDocument(data interface{}, path []string, indexes for i := 0; i < val.NumField(); i++ { field := typ.Field(i) fieldName := field.Name + // anonymous fields of type struct can elide the type name + if field.Anonymous && field.Type.Kind() == reflect.Struct { + fieldName = "" + } // if the field has a JSON name, prefer that jsonTag := field.Tag.Get("json") @@ -332,13 +336,18 @@ func (dm *DocumentMapping) walkDocument(data interface{}, path []string, indexes if jsonFieldName == "-" { continue } - if jsonFieldName != "" { + // allow json tag to set field name to empty, only if anonymous + if field.Tag != "" && (jsonFieldName != "" || field.Anonymous) { fieldName = jsonFieldName } if val.Field(i).CanInterface() { fieldVal := val.Field(i).Interface() - dm.processProperty(fieldVal, append(path, fieldName), indexes, context) + newpath := path + if fieldName != "" { + newpath = append(path, fieldName) + } + dm.processProperty(fieldVal, newpath, indexes, context) } } case reflect.Slice, reflect.Array: @@ -353,7 +362,18 @@ func (dm *DocumentMapping) walkDocument(data interface{}, path []string, indexes if ptrElem.IsValid() && ptrElem.CanInterface() { dm.processProperty(ptrElem.Interface(), path, indexes, context) } + case reflect.String: + dm.processProperty(val.String(), path, indexes, context) + case reflect.Int, reflect.Int8, reflect.Int16, reflect.Int32, reflect.Int64: + dm.processProperty(float64(val.Int()), path, indexes, context) + case reflect.Uint, reflect.Uint8, reflect.Uint16, reflect.Uint32, reflect.Uint64: + dm.processProperty(float64(val.Uint()), path, indexes, context) + case reflect.Float32, reflect.Float64: + dm.processProperty(float64(val.Float()), path, indexes, context) + case reflect.Bool: + dm.processProperty(val.Bool(), path, indexes, context) } + } func (dm *DocumentMapping) processProperty(property interface{}, path []string, indexes []uint64, context *walkContext) { diff --git a/vendor/github.com/blevesearch/bleve/mapping_index.go b/vendor/github.com/blevesearch/bleve/mapping_index.go index 20eac57..d91b075 100644 --- a/vendor/github.com/blevesearch/bleve/mapping_index.go +++ b/vendor/github.com/blevesearch/bleve/mapping_index.go @@ -15,7 +15,6 @@ import ( "github.com/blevesearch/bleve/analysis" "github.com/blevesearch/bleve/analysis/analyzers/standard_analyzer" - "github.com/blevesearch/bleve/analysis/byte_array_converters/json" "github.com/blevesearch/bleve/analysis/datetime_parsers/datetime_optional" "github.com/blevesearch/bleve/document" "github.com/blevesearch/bleve/registry" @@ -28,7 +27,6 @@ const defaultType = "_default" const defaultField = "_all" const defaultAnalyzer = standard_analyzer.Name const defaultDateTimeParser = datetime_optional.Name -const defaultByteArrayConverter = json_byte_array_converter.Name type customAnalysis struct { CharFilters map[string]map[string]interface{} `json:"char_filters,omitempty"` @@ -129,7 +127,6 @@ type IndexMapping struct { DefaultAnalyzer string `json:"default_analyzer"` DefaultDateTimeParser string `json:"default_datetime_parser"` DefaultField string `json:"default_field"` - ByteArrayConverter string `json:"byte_array_converter"` StoreDynamic bool `json:"store_dynamic"` IndexDynamic bool `json:"index_dynamic"` CustomAnalysis *customAnalysis `json:"analysis,omitempty"` @@ -234,7 +231,6 @@ func NewIndexMapping() *IndexMapping { DefaultAnalyzer: defaultAnalyzer, DefaultDateTimeParser: defaultDateTimeParser, DefaultField: defaultField, - ByteArrayConverter: defaultByteArrayConverter, IndexDynamic: IndexDynamic, StoreDynamic: StoreDynamic, CustomAnalysis: newCustomAnalysis(), @@ -296,7 +292,6 @@ func (im *IndexMapping) UnmarshalJSON(data []byte) error { im.DefaultAnalyzer = defaultAnalyzer im.DefaultDateTimeParser = defaultDateTimeParser im.DefaultField = defaultField - im.ByteArrayConverter = defaultByteArrayConverter im.DefaultMapping = NewDocumentMapping() im.TypeMapping = make(map[string]*DocumentMapping) im.StoreDynamic = StoreDynamic @@ -335,11 +330,6 @@ func (im *IndexMapping) UnmarshalJSON(data []byte) error { if err != nil { return err } - case "byte_array_converter": - err := json.Unmarshal(v, &im.ByteArrayConverter) - if err != nil { - return err - } case "default_mapping": err := json.Unmarshal(v, &im.DefaultMapping) if err != nil { @@ -394,26 +384,6 @@ func (im *IndexMapping) determineType(data interface{}) string { } func (im *IndexMapping) mapDocument(doc *document.Document, data interface{}) error { - // see if the top level object is a byte array, and possibly run through a converter - byteArrayData, ok := data.([]byte) - if ok { - byteArrayConverterConstructor := registry.ByteArrayConverterByName(im.ByteArrayConverter) - if byteArrayConverterConstructor != nil { - byteArrayConverter, err := byteArrayConverterConstructor(nil, nil) - if err == nil { - convertedData, err := byteArrayConverter.Convert(byteArrayData) - if err != nil { - return err - } - data = convertedData - } else { - logger.Printf("error creating byte array converter: %v", err) - } - } else { - logger.Printf("no byte array converter named: %s", im.ByteArrayConverter) - } - } - docType := im.determineType(data) docMapping := im.mappingForType(docType) walkContext := im.newWalkContext(doc, docMapping) diff --git a/vendor/github.com/blevesearch/bleve/numeric_util/prefix_coded.go b/vendor/github.com/blevesearch/bleve/numeric_util/prefix_coded.go index 2593115..4cf36de 100644 --- a/vendor/github.com/blevesearch/bleve/numeric_util/prefix_coded.go +++ b/vendor/github.com/blevesearch/bleve/numeric_util/prefix_coded.go @@ -9,9 +9,7 @@ package numeric_util -import ( - "fmt" -) +import "fmt" const ShiftStartInt64 byte = 0x20 @@ -72,3 +70,18 @@ func (p PrefixCoded) Int64() (int64, error) { } return int64(uint64((sortableBits << shift)) ^ 0x8000000000000000), nil } + +func ValidPrefixCodedTerm(p string) (bool, int) { + if len(p) > 0 { + if p[0] < ShiftStartInt64 || p[0] > ShiftStartInt64+63 { + return false, 0 + } + shift := p[0] - ShiftStartInt64 + nChars := ((63 - int(shift)) / 7) + 1 + if len(p) != nChars+1 { + return false, 0 + } + return true, int(shift) + } + return false, 0 +} diff --git a/vendor/github.com/blevesearch/bleve/query.go b/vendor/github.com/blevesearch/bleve/query.go index 724b7bd..595c783 100644 --- a/vendor/github.com/blevesearch/bleve/query.go +++ b/vendor/github.com/blevesearch/bleve/query.go @@ -271,7 +271,7 @@ func expandQuery(m *IndexMapping, query Query) (Query, error) { switch query.(type) { case *queryStringQuery: q := query.(*queryStringQuery) - parsed, err := parseQuerySyntax(q.Query, m) + parsed, err := parseQuerySyntax(q.Query) if err != nil { return nil, fmt.Errorf("could not parse '%s': %s", q.Query, err) } diff --git a/vendor/github.com/blevesearch/bleve/query_boolean.go b/vendor/github.com/blevesearch/bleve/query_boolean.go index 16e5950..8dcb8ef 100644 --- a/vendor/github.com/blevesearch/bleve/query_boolean.go +++ b/vendor/github.com/blevesearch/bleve/query_boolean.go @@ -123,6 +123,11 @@ func (q *booleanQuery) Searcher(i index.IndexReader, m *IndexMapping, explain bo return nil, err } } + + if mustSearcher == nil && shouldSearcher != nil && mustNotSearcher == nil { + return shouldSearcher, nil + } + return searchers.NewBooleanSearcher(i, mustSearcher, shouldSearcher, mustNotSearcher, explain) } diff --git a/vendor/github.com/blevesearch/bleve/query_conjunction.go b/vendor/github.com/blevesearch/bleve/query_conjunction.go index 7501891..99ad71c 100644 --- a/vendor/github.com/blevesearch/bleve/query_conjunction.go +++ b/vendor/github.com/blevesearch/bleve/query_conjunction.go @@ -58,6 +58,12 @@ func (q *conjunctionQuery) Searcher(i index.IndexReader, m *IndexMapping, explai } func (q *conjunctionQuery) Validate() error { + for _, q := range q.Conjuncts { + err := q.Validate() + if err != nil { + return err + } + } return nil } diff --git a/vendor/github.com/blevesearch/bleve/query_date_range.go b/vendor/github.com/blevesearch/bleve/query_date_range.go index 8643d83..df0de33 100644 --- a/vendor/github.com/blevesearch/bleve/query_date_range.go +++ b/vendor/github.com/blevesearch/bleve/query_date_range.go @@ -26,12 +26,12 @@ type dateRangeQuery struct { InclusiveEnd *bool `json:"inclusive_end,omitempty"` FieldVal string `json:"field,omitempty"` BoostVal float64 `json:"boost,omitempty"` - DateTimeParser *string `json:"datetime_parser,omitempty"` } // NewDateRangeQuery creates a new Query for ranges // of date values. -// A DateTimeParser is chosen based on the field. +// Date strings are parsed using the DateTimeParser configured in the +// top-level config.QueryDateTimeParser // Either, but not both endpoints can be nil. func NewDateRangeQuery(start, end *string) *dateRangeQuery { return NewDateRangeInclusiveQuery(start, end, nil, nil) @@ -39,7 +39,8 @@ func NewDateRangeQuery(start, end *string) *dateRangeQuery { // NewDateRangeInclusiveQuery creates a new Query for ranges // of date values. -// A DateTimeParser is chosen based on the field. +// Date strings are parsed using the DateTimeParser configured in the +// top-level config.QueryDateTimeParser // Either, but not both endpoints can be nil. // startInclusive and endInclusive control inclusion of the endpoints. func NewDateRangeInclusiveQuery(start, end *string, startInclusive, endInclusive *bool) *dateRangeQuery { @@ -72,15 +73,9 @@ func (q *dateRangeQuery) SetField(f string) Query { func (q *dateRangeQuery) Searcher(i index.IndexReader, m *IndexMapping, explain bool) (search.Searcher, error) { - dateTimeParserName := "" - if q.DateTimeParser != nil { - dateTimeParserName = *q.DateTimeParser - } else { - dateTimeParserName = m.datetimeParserNameForPath(q.FieldVal) - } - dateTimeParser := m.dateTimeParserNamed(dateTimeParserName) - if dateTimeParser == nil { - return nil, fmt.Errorf("no datetime parser named '%s' registered", *q.DateTimeParser) + min, max, err := q.parseEndpoints() + if err != nil { + return nil, err } field := q.FieldVal @@ -88,30 +83,43 @@ func (q *dateRangeQuery) Searcher(i index.IndexReader, m *IndexMapping, explain field = m.DefaultField } + return searchers.NewNumericRangeSearcher(i, min, max, q.InclusiveStart, q.InclusiveEnd, field, q.BoostVal, explain) +} + +func (q *dateRangeQuery) parseEndpoints() (*float64, *float64, error) { + dateTimeParser, err := Config.Cache.DateTimeParserNamed(Config.QueryDateTimeParser) + if err != nil { + return nil, nil, err + } + // now parse the endpoints min := math.Inf(-1) max := math.Inf(1) if q.Start != nil && *q.Start != "" { startTime, err := dateTimeParser.ParseDateTime(*q.Start) if err != nil { - return nil, err + return nil, nil, err } min = numeric_util.Int64ToFloat64(startTime.UnixNano()) } if q.End != nil && *q.End != "" { endTime, err := dateTimeParser.ParseDateTime(*q.End) if err != nil { - return nil, err + return nil, nil, err } max = numeric_util.Int64ToFloat64(endTime.UnixNano()) } - return searchers.NewNumericRangeSearcher(i, &min, &max, q.InclusiveStart, q.InclusiveEnd, field, q.BoostVal, explain) + return &min, &max, nil } func (q *dateRangeQuery) Validate() error { if q.Start == nil && q.Start == q.End { return fmt.Errorf("must specify start or end") } + _, _, err := q.parseEndpoints() + if err != nil { + return err + } return nil } diff --git a/vendor/github.com/blevesearch/bleve/query_disjunction.go b/vendor/github.com/blevesearch/bleve/query_disjunction.go index 7f9faef..78e4bd9 100644 --- a/vendor/github.com/blevesearch/bleve/query_disjunction.go +++ b/vendor/github.com/blevesearch/bleve/query_disjunction.go @@ -81,6 +81,12 @@ func (q *disjunctionQuery) Validate() error { if int(q.MinVal) > len(q.Disjuncts) { return ErrorDisjunctionFewerThanMinClauses } + for _, q := range q.Disjuncts { + err := q.Validate() + if err != nil { + return err + } + } return nil } diff --git a/vendor/github.com/blevesearch/bleve/query_match.go b/vendor/github.com/blevesearch/bleve/query_match.go index 582dbd6..79ea3bc 100644 --- a/vendor/github.com/blevesearch/bleve/query_match.go +++ b/vendor/github.com/blevesearch/bleve/query_match.go @@ -10,6 +10,7 @@ package bleve import ( + "encoding/json" "fmt" "github.com/blevesearch/bleve/index" @@ -17,12 +18,58 @@ import ( ) type matchQuery struct { - Match string `json:"match"` - FieldVal string `json:"field,omitempty"` - Analyzer string `json:"analyzer,omitempty"` - BoostVal float64 `json:"boost,omitempty"` - PrefixVal int `json:"prefix_length"` - FuzzinessVal int `json:"fuzziness"` + Match string `json:"match"` + FieldVal string `json:"field,omitempty"` + Analyzer string `json:"analyzer,omitempty"` + BoostVal float64 `json:"boost,omitempty"` + PrefixVal int `json:"prefix_length"` + FuzzinessVal int `json:"fuzziness"` + OperatorVal MatchQueryOperator `json:"operator,omitempty"` +} + +type MatchQueryOperator int + +const ( + // Document must satisfy AT LEAST ONE of term searches. + MatchQueryOperatorOr = 0 + // Document must satisfy ALL of term searches. + MatchQueryOperatorAnd = 1 +) + +func (o MatchQueryOperator) MarshalJSON() ([]byte, error) { + switch o { + case MatchQueryOperatorOr: + return json.Marshal("or") + case MatchQueryOperatorAnd: + return json.Marshal("and") + default: + return nil, fmt.Errorf("cannot marshal match operator %d to JSON", o) + } +} + +func (o *MatchQueryOperator) UnmarshalJSON(data []byte) error { + var operatorString string + err := json.Unmarshal(data, &operatorString) + if err != nil { + return err + } + + switch operatorString { + case "or": + *o = MatchQueryOperatorOr + return nil + case "and": + *o = MatchQueryOperatorAnd + return nil + default: + return matchQueryOperatorUnmarshalError(operatorString) + } +} + +type matchQueryOperatorUnmarshalError string + +func (e matchQueryOperatorUnmarshalError) Error() string { + return fmt.Sprintf("cannot unmarshal match operator '%s' from JSON", e) } // NewMatchQuery creates a Query for matching text. @@ -33,8 +80,23 @@ type matchQuery struct { // must satisfy at least one of these term searches. func NewMatchQuery(match string) *matchQuery { return &matchQuery{ - Match: match, - BoostVal: 1.0, + Match: match, + BoostVal: 1.0, + OperatorVal: MatchQueryOperatorOr, + } +} + +// NewMatchQuery creates a Query for matching text. +// An Analyzer is chosen based on the field. +// Input text is analyzed using this analyzer. +// Token terms resulting from this analysis are +// used to perform term searches. Result documents +// must satisfy term searches according to given operator. +func NewMatchQueryOperator(match string, operator MatchQueryOperator) *matchQuery { + return &matchQuery{ + Match: match, + BoostVal: 1.0, + OperatorVal: operator, } } @@ -74,6 +136,15 @@ func (q *matchQuery) SetPrefix(p int) Query { return q } +func (q *matchQuery) Operator() MatchQueryOperator { + return q.OperatorVal +} + +func (q *matchQuery) SetOperator(operator MatchQueryOperator) Query { + q.OperatorVal = operator + return q +} + func (q *matchQuery) Searcher(i index.IndexReader, m *IndexMapping, explain bool) (search.Searcher, error) { field := q.FieldVal @@ -114,10 +185,22 @@ func (q *matchQuery) Searcher(i index.IndexReader, m *IndexMapping, explain bool } } - shouldQuery := NewDisjunctionQueryMin(tqs, 1). - SetBoost(q.BoostVal) + switch q.OperatorVal { + case MatchQueryOperatorOr: + shouldQuery := NewDisjunctionQueryMin(tqs, 1). + SetBoost(q.BoostVal) - return shouldQuery.Searcher(i, m, explain) + return shouldQuery.Searcher(i, m, explain) + + case MatchQueryOperatorAnd: + mustQuery := NewConjunctionQuery(tqs). + SetBoost(q.BoostVal) + + return mustQuery.Searcher(i, m, explain) + + default: + return nil, fmt.Errorf("unhandled operator %d", q.OperatorVal) + } } noneQuery := NewMatchNoneQuery() return noneQuery.Searcher(i, m, explain) diff --git a/vendor/github.com/blevesearch/bleve/query_match_none.go b/vendor/github.com/blevesearch/bleve/query_match_none.go index 9b4ea8a..b13ca6c 100644 --- a/vendor/github.com/blevesearch/bleve/query_match_none.go +++ b/vendor/github.com/blevesearch/bleve/query_match_none.go @@ -10,6 +10,8 @@ package bleve import ( + "encoding/json" + "github.com/blevesearch/bleve/index" "github.com/blevesearch/bleve/search" "github.com/blevesearch/bleve/search/searchers" @@ -51,3 +53,11 @@ func (q *matchNoneQuery) Field() string { func (q *matchNoneQuery) SetField(f string) Query { return q } + +func (q *matchNoneQuery) MarshalJSON() ([]byte, error) { + tmp := map[string]interface{}{ + "boost": q.BoostVal, + "match_none": map[string]interface{}{}, + } + return json.Marshal(tmp) +} diff --git a/vendor/github.com/blevesearch/bleve/query_string.go b/vendor/github.com/blevesearch/bleve/query_string.go index 15e077e..005191c 100644 --- a/vendor/github.com/blevesearch/bleve/query_string.go +++ b/vendor/github.com/blevesearch/bleve/query_string.go @@ -39,7 +39,7 @@ func (q *queryStringQuery) SetBoost(b float64) Query { } func (q *queryStringQuery) Searcher(i index.IndexReader, m *IndexMapping, explain bool) (search.Searcher, error) { - newQuery, err := parseQuerySyntax(q.Query, m) + newQuery, err := parseQuerySyntax(q.Query) if err != nil { return nil, err } @@ -47,7 +47,11 @@ func (q *queryStringQuery) Searcher(i index.IndexReader, m *IndexMapping, explai } func (q *queryStringQuery) Validate() error { - return nil + newQuery, err := parseQuerySyntax(q.Query) + if err != nil { + return err + } + return newQuery.Validate() } func (q *queryStringQuery) Field() string { diff --git a/vendor/github.com/blevesearch/bleve/query_string.nex b/vendor/github.com/blevesearch/bleve/query_string.nex deleted file mode 100644 index f501445..0000000 --- a/vendor/github.com/blevesearch/bleve/query_string.nex +++ /dev/null @@ -1,51 +0,0 @@ -/\"((\\\")|(\\\\)|(\\\/)|(\\b)|(\\f)|(\\n)|(\\r)|(\\t)|(\\u[0-9a-fA-F][0-9a-fA-F][0-9a-fA-F][0-9a-fA-F])|[^\"])*\"/ { - lval.s = yylex.Text()[1:len(yylex.Text())-1] - logDebugTokens("PHRASE - %s", lval.s); - return tPHRASE - } -/\/((\\\")|(\\\\)|(\\\/)|(\\b)|(\\f)|(\\n)|(\\r)|(\\t)|(\\u[0-9a-fA-F][0-9a-fA-F][0-9a-fA-F][0-9a-fA-F])|[^\/])*\// { - lval.s = yylex.Text()[1:len(yylex.Text())-1] - logDebugTokens("REGEXP - %s", lval.s); - return tREGEXP - } -/\+/ { logDebugTokens("PLUS"); return tPLUS } -/-/ { logDebugTokens("MINUS"); return tMINUS } -/:/ { logDebugTokens("COLON"); return tCOLON } -/\^/ { logDebugTokens("BOOST"); return tBOOST } -/\(/ { logDebugTokens("LPAREN"); return tLPAREN } -/\)/ { logDebugTokens("RPAREN"); return tRPAREN } -/>/ { logDebugTokens("GREATER"); return tGREATER } -/<=~-][^\t\n\f\r :^~\*\?]*/ { - lval.s = yylex.Text() - logDebugTokens("STRING - %s", lval.s); - return tSTRING - } -/[^\t\n\f\r :^\+\*\?><=~-][^\t\n\f\r :^~]*/ { - lval.s = yylex.Text() - logDebugTokens("WILD - %s", lval.s); - return tWILD - } -// -package bleve - -func logDebugTokens(format string, v ...interface{}) { - if debugLexer { - logger.Printf(format, v...) - } -} diff --git a/vendor/github.com/blevesearch/bleve/query_string.nn.go b/vendor/github.com/blevesearch/bleve/query_string.nn.go deleted file mode 100644 index 6592e07..0000000 --- a/vendor/github.com/blevesearch/bleve/query_string.nn.go +++ /dev/null @@ -1,2071 +0,0 @@ -package bleve - -import ( - "bufio" - "io" - "strings" -) - -type frame struct { - i int - s string - line, column int -} -type lexer struct { - // The lexer runs in its own goroutine, and communicates via channel 'ch'. - ch chan frame - // We record the level of nesting because the action could return, and a - // subsequent call expects to pick up where it left off. In other words, - // we're simulating a coroutine. - // TODO: Support a channel-based variant that compatible with Go's yacc. - stack []frame - stale bool - - // The 'l' and 'c' fields were added for - // https://github.com/wagerlabs/docker/blob/65694e801a7b80930961d70c69cba9f2465459be/buildfile.nex - // Since then, I introduced the built-in Line() and Column() functions. - l, c int - - parseResult interface{} - - // The following line makes it easy for scripts to insert fields in the - // generated code. - // [NEX_END_OF_LEXER_STRUCT] -} - -// newLexerWithInit creates a new lexer object, runs the given callback on it, -// then returns it. -func newLexerWithInit(in io.Reader, initFun func(*lexer)) *lexer { - type dfa struct { - acc []bool // Accepting states. - f []func(rune) int // Transitions. - startf, endf []int // Transitions at start and end of input. - nest []dfa - } - yylex := new(lexer) - if initFun != nil { - initFun(yylex) - } - yylex.ch = make(chan frame) - var scan func(in *bufio.Reader, ch chan frame, family []dfa, line, column int) - scan = func(in *bufio.Reader, ch chan frame, family []dfa, line, column int) { - // Index of DFA and length of highest-precedence match so far. - matchi, matchn := 0, -1 - var buf []rune - n := 0 - checkAccept := func(i int, st int) bool { - // Higher precedence match? DFAs are run in parallel, so matchn is at most len(buf), hence we may omit the length equality check. - if family[i].acc[st] && (matchn < n || matchi > i) { - matchi, matchn = i, n - return true - } - return false - } - var state [][2]int - for i := 0; i < len(family); i++ { - mark := make([]bool, len(family[i].startf)) - // Every DFA starts at state 0. - st := 0 - for { - state = append(state, [2]int{i, st}) - mark[st] = true - // As we're at the start of input, follow all ^ transitions and append to our list of start states. - st = family[i].startf[st] - if -1 == st || mark[st] { - break - } - // We only check for a match after at least one transition. - checkAccept(i, st) - } - } - atEOF := false - for { - if n == len(buf) && !atEOF { - r, _, err := in.ReadRune() - switch err { - case io.EOF: - atEOF = true - case nil: - buf = append(buf, r) - default: - panic(err) - } - } - if !atEOF { - r := buf[n] - n++ - var nextState [][2]int - for _, x := range state { - x[1] = family[x[0]].f[x[1]](r) - if -1 == x[1] { - continue - } - nextState = append(nextState, x) - checkAccept(x[0], x[1]) - } - state = nextState - } else { - dollar: // Handle $. - for _, x := range state { - mark := make([]bool, len(family[x[0]].endf)) - for { - mark[x[1]] = true - x[1] = family[x[0]].endf[x[1]] - if -1 == x[1] || mark[x[1]] { - break - } - if checkAccept(x[0], x[1]) { - // Unlike before, we can break off the search. Now that we're at the end, there's no need to maintain the state of each DFA. - break dollar - } - } - } - state = nil - } - - if state == nil { - lcUpdate := func(r rune) { - if r == '\n' { - line++ - column = 0 - } else { - column++ - } - } - // All DFAs stuck. Return last match if it exists, otherwise advance by one rune and restart all DFAs. - if matchn == -1 { - if len(buf) == 0 { // This can only happen at the end of input. - break - } - lcUpdate(buf[0]) - buf = buf[1:] - } else { - text := string(buf[:matchn]) - buf = buf[matchn:] - matchn = -1 - ch <- frame{matchi, text, line, column} - if len(family[matchi].nest) > 0 { - scan(bufio.NewReader(strings.NewReader(text)), ch, family[matchi].nest, line, column) - } - if atEOF { - break - } - for _, r := range text { - lcUpdate(r) - } - } - n = 0 - for i := 0; i < len(family); i++ { - state = append(state, [2]int{i, 0}) - } - } - } - ch <- frame{-1, "", line, column} - } - go scan(bufio.NewReader(in), yylex.ch, []dfa{ - // \"((\\\")|(\\\\)|(\\\/)|(\\b)|(\\f)|(\\n)|(\\r)|(\\t)|(\\u[0-9a-fA-F][0-9a-fA-F][0-9a-fA-F][0-9a-fA-F])|[^\"])*\" - {[]bool{false, false, true, false, false, true, false, false, false, false, false, false, false, false, false, false, false, false}, []func(rune) int{ // Transitions - func(r rune) int { - switch r { - case 34: - return 1 - case 47: - return -1 - case 92: - return -1 - case 98: - return -1 - case 102: - return -1 - case 110: - return -1 - case 114: - return -1 - case 116: - return -1 - case 117: - return -1 - } - switch { - case 48 <= r && r <= 57: - return -1 - case 65 <= r && r <= 70: - return -1 - case 97 <= r && r <= 102: - return -1 - } - return -1 - }, - func(r rune) int { - switch r { - case 34: - return 2 - case 47: - return 3 - case 92: - return 4 - case 98: - return 3 - case 102: - return 3 - case 110: - return 3 - case 114: - return 3 - case 116: - return 3 - case 117: - return 3 - } - switch { - case 48 <= r && r <= 57: - return 3 - case 65 <= r && r <= 70: - return 3 - case 97 <= r && r <= 102: - return 3 - } - return 3 - }, - func(r rune) int { - switch r { - case 34: - return -1 - case 47: - return -1 - case 92: - return -1 - case 98: - return -1 - case 102: - return -1 - case 110: - return -1 - case 114: - return -1 - case 116: - return -1 - case 117: - return -1 - } - switch { - case 48 <= r && r <= 57: - return -1 - case 65 <= r && r <= 70: - return -1 - case 97 <= r && r <= 102: - return -1 - } - return -1 - }, - func(r rune) int { - switch r { - case 34: - return 2 - case 47: - return 3 - case 92: - return 4 - case 98: - return 3 - case 102: - return 3 - case 110: - return 3 - case 114: - return 3 - case 116: - return 3 - case 117: - return 3 - } - switch { - case 48 <= r && r <= 57: - return 3 - case 65 <= r && r <= 70: - return 3 - case 97 <= r && r <= 102: - return 3 - } - return 3 - }, - func(r rune) int { - switch r { - case 34: - return 5 - case 47: - return 6 - case 92: - return 7 - case 98: - return 8 - case 102: - return 9 - case 110: - return 10 - case 114: - return 11 - case 116: - return 12 - case 117: - return 13 - } - switch { - case 48 <= r && r <= 57: - return 3 - case 65 <= r && r <= 70: - return 3 - case 97 <= r && r <= 102: - return 3 - } - return 3 - }, - func(r rune) int { - switch r { - case 34: - return 2 - case 47: - return 3 - case 92: - return 4 - case 98: - return 3 - case 102: - return 3 - case 110: - return 3 - case 114: - return 3 - case 116: - return 3 - case 117: - return 3 - } - switch { - case 48 <= r && r <= 57: - return 3 - case 65 <= r && r <= 70: - return 3 - case 97 <= r && r <= 102: - return 3 - } - return 3 - }, - func(r rune) int { - switch r { - case 34: - return 2 - case 47: - return 3 - case 92: - return 4 - case 98: - return 3 - case 102: - return 3 - case 110: - return 3 - case 114: - return 3 - case 116: - return 3 - case 117: - return 3 - } - switch { - case 48 <= r && r <= 57: - return 3 - case 65 <= r && r <= 70: - return 3 - case 97 <= r && r <= 102: - return 3 - } - return 3 - }, - func(r rune) int { - switch r { - case 34: - return 5 - case 47: - return 6 - case 92: - return 7 - case 98: - return 8 - case 102: - return 9 - case 110: - return 10 - case 114: - return 11 - case 116: - return 12 - case 117: - return 13 - } - switch { - case 48 <= r && r <= 57: - return 3 - case 65 <= r && r <= 70: - return 3 - case 97 <= r && r <= 102: - return 3 - } - return 3 - }, - func(r rune) int { - switch r { - case 34: - return 2 - case 47: - return 3 - case 92: - return 4 - case 98: - return 3 - case 102: - return 3 - case 110: - return 3 - case 114: - return 3 - case 116: - return 3 - case 117: - return 3 - } - switch { - case 48 <= r && r <= 57: - return 3 - case 65 <= r && r <= 70: - return 3 - case 97 <= r && r <= 102: - return 3 - } - return 3 - }, - func(r rune) int { - switch r { - case 34: - return 2 - case 47: - return 3 - case 92: - return 4 - case 98: - return 3 - case 102: - return 3 - case 110: - return 3 - case 114: - return 3 - case 116: - return 3 - case 117: - return 3 - } - switch { - case 48 <= r && r <= 57: - return 3 - case 65 <= r && r <= 70: - return 3 - case 97 <= r && r <= 102: - return 3 - } - return 3 - }, - func(r rune) int { - switch r { - case 34: - return 2 - case 47: - return 3 - case 92: - return 4 - case 98: - return 3 - case 102: - return 3 - case 110: - return 3 - case 114: - return 3 - case 116: - return 3 - case 117: - return 3 - } - switch { - case 48 <= r && r <= 57: - return 3 - case 65 <= r && r <= 70: - return 3 - case 97 <= r && r <= 102: - return 3 - } - return 3 - }, - func(r rune) int { - switch r { - case 34: - return 2 - case 47: - return 3 - case 92: - return 4 - case 98: - return 3 - case 102: - return 3 - case 110: - return 3 - case 114: - return 3 - case 116: - return 3 - case 117: - return 3 - } - switch { - case 48 <= r && r <= 57: - return 3 - case 65 <= r && r <= 70: - return 3 - case 97 <= r && r <= 102: - return 3 - } - return 3 - }, - func(r rune) int { - switch r { - case 34: - return 2 - case 47: - return 3 - case 92: - return 4 - case 98: - return 3 - case 102: - return 3 - case 110: - return 3 - case 114: - return 3 - case 116: - return 3 - case 117: - return 3 - } - switch { - case 48 <= r && r <= 57: - return 3 - case 65 <= r && r <= 70: - return 3 - case 97 <= r && r <= 102: - return 3 - } - return 3 - }, - func(r rune) int { - switch r { - case 34: - return 2 - case 47: - return 3 - case 92: - return 4 - case 98: - return 14 - case 102: - return 14 - case 110: - return 3 - case 114: - return 3 - case 116: - return 3 - case 117: - return 3 - } - switch { - case 48 <= r && r <= 57: - return 14 - case 65 <= r && r <= 70: - return 14 - case 97 <= r && r <= 102: - return 14 - } - return 3 - }, - func(r rune) int { - switch r { - case 34: - return 2 - case 47: - return 3 - case 92: - return 4 - case 98: - return 15 - case 102: - return 15 - case 110: - return 3 - case 114: - return 3 - case 116: - return 3 - case 117: - return 3 - } - switch { - case 48 <= r && r <= 57: - return 15 - case 65 <= r && r <= 70: - return 15 - case 97 <= r && r <= 102: - return 15 - } - return 3 - }, - func(r rune) int { - switch r { - case 34: - return 2 - case 47: - return 3 - case 92: - return 4 - case 98: - return 16 - case 102: - return 16 - case 110: - return 3 - case 114: - return 3 - case 116: - return 3 - case 117: - return 3 - } - switch { - case 48 <= r && r <= 57: - return 16 - case 65 <= r && r <= 70: - return 16 - case 97 <= r && r <= 102: - return 16 - } - return 3 - }, - func(r rune) int { - switch r { - case 34: - return 2 - case 47: - return 3 - case 92: - return 4 - case 98: - return 17 - case 102: - return 17 - case 110: - return 3 - case 114: - return 3 - case 116: - return 3 - case 117: - return 3 - } - switch { - case 48 <= r && r <= 57: - return 17 - case 65 <= r && r <= 70: - return 17 - case 97 <= r && r <= 102: - return 17 - } - return 3 - }, - func(r rune) int { - switch r { - case 34: - return 2 - case 47: - return 3 - case 92: - return 4 - case 98: - return 3 - case 102: - return 3 - case 110: - return 3 - case 114: - return 3 - case 116: - return 3 - case 117: - return 3 - } - switch { - case 48 <= r && r <= 57: - return 3 - case 65 <= r && r <= 70: - return 3 - case 97 <= r && r <= 102: - return 3 - } - return 3 - }, - }, []int{ /* Start-of-input transitions */ -1, -1, -1, -1, -1, -1, -1, -1, -1, -1, -1, -1, -1, -1, -1, -1, -1, -1}, []int{ /* End-of-input transitions */ -1, -1, -1, -1, -1, -1, -1, -1, -1, -1, -1, -1, -1, -1, -1, -1, -1, -1}, nil}, - - // \/((\\\")|(\\\\)|(\\\/)|(\\b)|(\\f)|(\\n)|(\\r)|(\\t)|(\\u[0-9a-fA-F][0-9a-fA-F][0-9a-fA-F][0-9a-fA-F])|[^\/])*\/ - {[]bool{false, false, false, true, false, false, true, false, false, false, false, false, false, false, false, false, false, false}, []func(rune) int{ // Transitions - func(r rune) int { - switch r { - case 34: - return -1 - case 47: - return 1 - case 92: - return -1 - case 98: - return -1 - case 102: - return -1 - case 110: - return -1 - case 114: - return -1 - case 116: - return -1 - case 117: - return -1 - } - switch { - case 48 <= r && r <= 57: - return -1 - case 65 <= r && r <= 70: - return -1 - case 97 <= r && r <= 102: - return -1 - } - return -1 - }, - func(r rune) int { - switch r { - case 34: - return 2 - case 47: - return 3 - case 92: - return 4 - case 98: - return 2 - case 102: - return 2 - case 110: - return 2 - case 114: - return 2 - case 116: - return 2 - case 117: - return 2 - } - switch { - case 48 <= r && r <= 57: - return 2 - case 65 <= r && r <= 70: - return 2 - case 97 <= r && r <= 102: - return 2 - } - return 2 - }, - func(r rune) int { - switch r { - case 34: - return 2 - case 47: - return 3 - case 92: - return 4 - case 98: - return 2 - case 102: - return 2 - case 110: - return 2 - case 114: - return 2 - case 116: - return 2 - case 117: - return 2 - } - switch { - case 48 <= r && r <= 57: - return 2 - case 65 <= r && r <= 70: - return 2 - case 97 <= r && r <= 102: - return 2 - } - return 2 - }, - func(r rune) int { - switch r { - case 34: - return -1 - case 47: - return -1 - case 92: - return -1 - case 98: - return -1 - case 102: - return -1 - case 110: - return -1 - case 114: - return -1 - case 116: - return -1 - case 117: - return -1 - } - switch { - case 48 <= r && r <= 57: - return -1 - case 65 <= r && r <= 70: - return -1 - case 97 <= r && r <= 102: - return -1 - } - return -1 - }, - func(r rune) int { - switch r { - case 34: - return 5 - case 47: - return 6 - case 92: - return 7 - case 98: - return 8 - case 102: - return 9 - case 110: - return 10 - case 114: - return 11 - case 116: - return 12 - case 117: - return 13 - } - switch { - case 48 <= r && r <= 57: - return 2 - case 65 <= r && r <= 70: - return 2 - case 97 <= r && r <= 102: - return 2 - } - return 2 - }, - func(r rune) int { - switch r { - case 34: - return 2 - case 47: - return 3 - case 92: - return 4 - case 98: - return 2 - case 102: - return 2 - case 110: - return 2 - case 114: - return 2 - case 116: - return 2 - case 117: - return 2 - } - switch { - case 48 <= r && r <= 57: - return 2 - case 65 <= r && r <= 70: - return 2 - case 97 <= r && r <= 102: - return 2 - } - return 2 - }, - func(r rune) int { - switch r { - case 34: - return 2 - case 47: - return 3 - case 92: - return 4 - case 98: - return 2 - case 102: - return 2 - case 110: - return 2 - case 114: - return 2 - case 116: - return 2 - case 117: - return 2 - } - switch { - case 48 <= r && r <= 57: - return 2 - case 65 <= r && r <= 70: - return 2 - case 97 <= r && r <= 102: - return 2 - } - return 2 - }, - func(r rune) int { - switch r { - case 34: - return 5 - case 47: - return 6 - case 92: - return 7 - case 98: - return 8 - case 102: - return 9 - case 110: - return 10 - case 114: - return 11 - case 116: - return 12 - case 117: - return 13 - } - switch { - case 48 <= r && r <= 57: - return 2 - case 65 <= r && r <= 70: - return 2 - case 97 <= r && r <= 102: - return 2 - } - return 2 - }, - func(r rune) int { - switch r { - case 34: - return 2 - case 47: - return 3 - case 92: - return 4 - case 98: - return 2 - case 102: - return 2 - case 110: - return 2 - case 114: - return 2 - case 116: - return 2 - case 117: - return 2 - } - switch { - case 48 <= r && r <= 57: - return 2 - case 65 <= r && r <= 70: - return 2 - case 97 <= r && r <= 102: - return 2 - } - return 2 - }, - func(r rune) int { - switch r { - case 34: - return 2 - case 47: - return 3 - case 92: - return 4 - case 98: - return 2 - case 102: - return 2 - case 110: - return 2 - case 114: - return 2 - case 116: - return 2 - case 117: - return 2 - } - switch { - case 48 <= r && r <= 57: - return 2 - case 65 <= r && r <= 70: - return 2 - case 97 <= r && r <= 102: - return 2 - } - return 2 - }, - func(r rune) int { - switch r { - case 34: - return 2 - case 47: - return 3 - case 92: - return 4 - case 98: - return 2 - case 102: - return 2 - case 110: - return 2 - case 114: - return 2 - case 116: - return 2 - case 117: - return 2 - } - switch { - case 48 <= r && r <= 57: - return 2 - case 65 <= r && r <= 70: - return 2 - case 97 <= r && r <= 102: - return 2 - } - return 2 - }, - func(r rune) int { - switch r { - case 34: - return 2 - case 47: - return 3 - case 92: - return 4 - case 98: - return 2 - case 102: - return 2 - case 110: - return 2 - case 114: - return 2 - case 116: - return 2 - case 117: - return 2 - } - switch { - case 48 <= r && r <= 57: - return 2 - case 65 <= r && r <= 70: - return 2 - case 97 <= r && r <= 102: - return 2 - } - return 2 - }, - func(r rune) int { - switch r { - case 34: - return 2 - case 47: - return 3 - case 92: - return 4 - case 98: - return 2 - case 102: - return 2 - case 110: - return 2 - case 114: - return 2 - case 116: - return 2 - case 117: - return 2 - } - switch { - case 48 <= r && r <= 57: - return 2 - case 65 <= r && r <= 70: - return 2 - case 97 <= r && r <= 102: - return 2 - } - return 2 - }, - func(r rune) int { - switch r { - case 34: - return 2 - case 47: - return 3 - case 92: - return 4 - case 98: - return 14 - case 102: - return 14 - case 110: - return 2 - case 114: - return 2 - case 116: - return 2 - case 117: - return 2 - } - switch { - case 48 <= r && r <= 57: - return 14 - case 65 <= r && r <= 70: - return 14 - case 97 <= r && r <= 102: - return 14 - } - return 2 - }, - func(r rune) int { - switch r { - case 34: - return 2 - case 47: - return 3 - case 92: - return 4 - case 98: - return 15 - case 102: - return 15 - case 110: - return 2 - case 114: - return 2 - case 116: - return 2 - case 117: - return 2 - } - switch { - case 48 <= r && r <= 57: - return 15 - case 65 <= r && r <= 70: - return 15 - case 97 <= r && r <= 102: - return 15 - } - return 2 - }, - func(r rune) int { - switch r { - case 34: - return 2 - case 47: - return 3 - case 92: - return 4 - case 98: - return 16 - case 102: - return 16 - case 110: - return 2 - case 114: - return 2 - case 116: - return 2 - case 117: - return 2 - } - switch { - case 48 <= r && r <= 57: - return 16 - case 65 <= r && r <= 70: - return 16 - case 97 <= r && r <= 102: - return 16 - } - return 2 - }, - func(r rune) int { - switch r { - case 34: - return 2 - case 47: - return 3 - case 92: - return 4 - case 98: - return 17 - case 102: - return 17 - case 110: - return 2 - case 114: - return 2 - case 116: - return 2 - case 117: - return 2 - } - switch { - case 48 <= r && r <= 57: - return 17 - case 65 <= r && r <= 70: - return 17 - case 97 <= r && r <= 102: - return 17 - } - return 2 - }, - func(r rune) int { - switch r { - case 34: - return 2 - case 47: - return 3 - case 92: - return 4 - case 98: - return 2 - case 102: - return 2 - case 110: - return 2 - case 114: - return 2 - case 116: - return 2 - case 117: - return 2 - } - switch { - case 48 <= r && r <= 57: - return 2 - case 65 <= r && r <= 70: - return 2 - case 97 <= r && r <= 102: - return 2 - } - return 2 - }, - }, []int{ /* Start-of-input transitions */ -1, -1, -1, -1, -1, -1, -1, -1, -1, -1, -1, -1, -1, -1, -1, -1, -1, -1}, []int{ /* End-of-input transitions */ -1, -1, -1, -1, -1, -1, -1, -1, -1, -1, -1, -1, -1, -1, -1, -1, -1, -1}, nil}, - - // \+ - {[]bool{false, true}, []func(rune) int{ // Transitions - func(r rune) int { - switch r { - case 43: - return 1 - } - return -1 - }, - func(r rune) int { - switch r { - case 43: - return -1 - } - return -1 - }, - }, []int{ /* Start-of-input transitions */ -1, -1}, []int{ /* End-of-input transitions */ -1, -1}, nil}, - - // - - {[]bool{false, true}, []func(rune) int{ // Transitions - func(r rune) int { - switch r { - case 45: - return 1 - } - return -1 - }, - func(r rune) int { - switch r { - case 45: - return -1 - } - return -1 - }, - }, []int{ /* Start-of-input transitions */ -1, -1}, []int{ /* End-of-input transitions */ -1, -1}, nil}, - - // : - {[]bool{false, true}, []func(rune) int{ // Transitions - func(r rune) int { - switch r { - case 58: - return 1 - } - return -1 - }, - func(r rune) int { - switch r { - case 58: - return -1 - } - return -1 - }, - }, []int{ /* Start-of-input transitions */ -1, -1}, []int{ /* End-of-input transitions */ -1, -1}, nil}, - - // \^ - {[]bool{false, true}, []func(rune) int{ // Transitions - func(r rune) int { - switch r { - case 94: - return 1 - } - return -1 - }, - func(r rune) int { - switch r { - case 94: - return -1 - } - return -1 - }, - }, []int{ /* Start-of-input transitions */ -1, -1}, []int{ /* End-of-input transitions */ -1, -1}, nil}, - - // \( - {[]bool{false, true}, []func(rune) int{ // Transitions - func(r rune) int { - switch r { - case 40: - return 1 - } - return -1 - }, - func(r rune) int { - switch r { - case 40: - return -1 - } - return -1 - }, - }, []int{ /* Start-of-input transitions */ -1, -1}, []int{ /* End-of-input transitions */ -1, -1}, nil}, - - // \) - {[]bool{false, true}, []func(rune) int{ // Transitions - func(r rune) int { - switch r { - case 41: - return 1 - } - return -1 - }, - func(r rune) int { - switch r { - case 41: - return -1 - } - return -1 - }, - }, []int{ /* Start-of-input transitions */ -1, -1}, []int{ /* End-of-input transitions */ -1, -1}, nil}, - - // > - {[]bool{false, true}, []func(rune) int{ // Transitions - func(r rune) int { - switch r { - case 62: - return 1 - } - return -1 - }, - func(r rune) int { - switch r { - case 62: - return -1 - } - return -1 - }, - }, []int{ /* Start-of-input transitions */ -1, -1}, []int{ /* End-of-input transitions */ -1, -1}, nil}, - - // < - {[]bool{false, true}, []func(rune) int{ // Transitions - func(r rune) int { - switch r { - case 60: - return 1 - } - return -1 - }, - func(r rune) int { - switch r { - case 60: - return -1 - } - return -1 - }, - }, []int{ /* Start-of-input transitions */ -1, -1}, []int{ /* End-of-input transitions */ -1, -1}, nil}, - - // = - {[]bool{false, true}, []func(rune) int{ // Transitions - func(r rune) int { - switch r { - case 61: - return 1 - } - return -1 - }, - func(r rune) int { - switch r { - case 61: - return -1 - } - return -1 - }, - }, []int{ /* Start-of-input transitions */ -1, -1}, []int{ /* End-of-input transitions */ -1, -1}, nil}, - - // ~([0-9]|[1-9][0-9]*) - {[]bool{false, false, true, true, true}, []func(rune) int{ // Transitions - func(r rune) int { - switch r { - case 126: - return 1 - } - switch { - case 48 <= r && r <= 48: - return -1 - case 49 <= r && r <= 57: - return -1 - } - return -1 - }, - func(r rune) int { - switch r { - case 126: - return -1 - } - switch { - case 48 <= r && r <= 48: - return 2 - case 49 <= r && r <= 57: - return 3 - } - return -1 - }, - func(r rune) int { - switch r { - case 126: - return -1 - } - switch { - case 48 <= r && r <= 48: - return -1 - case 49 <= r && r <= 57: - return -1 - } - return -1 - }, - func(r rune) int { - switch r { - case 126: - return -1 - } - switch { - case 48 <= r && r <= 48: - return 4 - case 49 <= r && r <= 57: - return 4 - } - return -1 - }, - func(r rune) int { - switch r { - case 126: - return -1 - } - switch { - case 48 <= r && r <= 48: - return 4 - case 49 <= r && r <= 57: - return 4 - } - return -1 - }, - }, []int{ /* Start-of-input transitions */ -1, -1, -1, -1, -1}, []int{ /* End-of-input transitions */ -1, -1, -1, -1, -1}, nil}, - - // ~ - {[]bool{false, true}, []func(rune) int{ // Transitions - func(r rune) int { - switch r { - case 126: - return 1 - } - return -1 - }, - func(r rune) int { - switch r { - case 126: - return -1 - } - return -1 - }, - }, []int{ /* Start-of-input transitions */ -1, -1}, []int{ /* End-of-input transitions */ -1, -1}, nil}, - - // -?([0-9]|[1-9][0-9]*)(\.[0-9][0-9]*)? - {[]bool{false, false, true, true, false, true, true, true}, []func(rune) int{ // Transitions - func(r rune) int { - switch r { - case 45: - return 1 - case 46: - return -1 - } - switch { - case 48 <= r && r <= 48: - return 2 - case 49 <= r && r <= 57: - return 3 - } - return -1 - }, - func(r rune) int { - switch r { - case 45: - return -1 - case 46: - return -1 - } - switch { - case 48 <= r && r <= 48: - return 2 - case 49 <= r && r <= 57: - return 3 - } - return -1 - }, - func(r rune) int { - switch r { - case 45: - return -1 - case 46: - return 4 - } - switch { - case 48 <= r && r <= 48: - return -1 - case 49 <= r && r <= 57: - return -1 - } - return -1 - }, - func(r rune) int { - switch r { - case 45: - return -1 - case 46: - return 4 - } - switch { - case 48 <= r && r <= 48: - return 5 - case 49 <= r && r <= 57: - return 5 - } - return -1 - }, - func(r rune) int { - switch r { - case 45: - return -1 - case 46: - return -1 - } - switch { - case 48 <= r && r <= 48: - return 6 - case 49 <= r && r <= 57: - return 6 - } - return -1 - }, - func(r rune) int { - switch r { - case 45: - return -1 - case 46: - return 4 - } - switch { - case 48 <= r && r <= 48: - return 5 - case 49 <= r && r <= 57: - return 5 - } - return -1 - }, - func(r rune) int { - switch r { - case 45: - return -1 - case 46: - return -1 - } - switch { - case 48 <= r && r <= 48: - return 7 - case 49 <= r && r <= 57: - return 7 - } - return -1 - }, - func(r rune) int { - switch r { - case 45: - return -1 - case 46: - return -1 - } - switch { - case 48 <= r && r <= 48: - return 7 - case 49 <= r && r <= 57: - return 7 - } - return -1 - }, - }, []int{ /* Start-of-input transitions */ -1, -1, -1, -1, -1, -1, -1, -1}, []int{ /* End-of-input transitions */ -1, -1, -1, -1, -1, -1, -1, -1}, nil}, - - // [ \t\n]+ - {[]bool{false, true}, []func(rune) int{ // Transitions - func(r rune) int { - switch r { - case 9: - return 1 - case 10: - return 1 - case 32: - return 1 - } - return -1 - }, - func(r rune) int { - switch r { - case 9: - return 1 - case 10: - return 1 - case 32: - return 1 - } - return -1 - }, - }, []int{ /* Start-of-input transitions */ -1, -1}, []int{ /* End-of-input transitions */ -1, -1}, nil}, - - // [^\t\n\f\r :^\+\*\?><=~-][^\t\n\f\r :^~\*\?]* - {[]bool{false, true, true}, []func(rune) int{ // Transitions - func(r rune) int { - switch r { - case 9: - return -1 - case 10: - return -1 - case 12: - return -1 - case 13: - return -1 - case 32: - return -1 - case 42: - return -1 - case 43: - return -1 - case 45: - return -1 - case 58: - return -1 - case 60: - return -1 - case 61: - return -1 - case 62: - return -1 - case 63: - return -1 - case 94: - return -1 - case 126: - return -1 - } - return 1 - }, - func(r rune) int { - switch r { - case 9: - return -1 - case 10: - return -1 - case 12: - return -1 - case 13: - return -1 - case 32: - return -1 - case 42: - return -1 - case 43: - return 2 - case 45: - return 2 - case 58: - return -1 - case 60: - return 2 - case 61: - return 2 - case 62: - return 2 - case 63: - return -1 - case 94: - return -1 - case 126: - return -1 - } - return 2 - }, - func(r rune) int { - switch r { - case 9: - return -1 - case 10: - return -1 - case 12: - return -1 - case 13: - return -1 - case 32: - return -1 - case 42: - return -1 - case 43: - return 2 - case 45: - return 2 - case 58: - return -1 - case 60: - return 2 - case 61: - return 2 - case 62: - return 2 - case 63: - return -1 - case 94: - return -1 - case 126: - return -1 - } - return 2 - }, - }, []int{ /* Start-of-input transitions */ -1, -1, -1}, []int{ /* End-of-input transitions */ -1, -1, -1}, nil}, - - // [^\t\n\f\r :^\+\*\?><=~-][^\t\n\f\r :^~]* - {[]bool{false, true, true}, []func(rune) int{ // Transitions - func(r rune) int { - switch r { - case 9: - return -1 - case 10: - return -1 - case 12: - return -1 - case 13: - return -1 - case 32: - return -1 - case 42: - return -1 - case 43: - return -1 - case 45: - return -1 - case 58: - return -1 - case 60: - return -1 - case 61: - return -1 - case 62: - return -1 - case 63: - return -1 - case 94: - return -1 - case 126: - return -1 - } - return 1 - }, - func(r rune) int { - switch r { - case 9: - return -1 - case 10: - return -1 - case 12: - return -1 - case 13: - return -1 - case 32: - return -1 - case 42: - return 2 - case 43: - return 2 - case 45: - return 2 - case 58: - return -1 - case 60: - return 2 - case 61: - return 2 - case 62: - return 2 - case 63: - return 2 - case 94: - return -1 - case 126: - return -1 - } - return 2 - }, - func(r rune) int { - switch r { - case 9: - return -1 - case 10: - return -1 - case 12: - return -1 - case 13: - return -1 - case 32: - return -1 - case 42: - return 2 - case 43: - return 2 - case 45: - return 2 - case 58: - return -1 - case 60: - return 2 - case 61: - return 2 - case 62: - return 2 - case 63: - return 2 - case 94: - return -1 - case 126: - return -1 - } - return 2 - }, - }, []int{ /* Start-of-input transitions */ -1, -1, -1}, []int{ /* End-of-input transitions */ -1, -1, -1}, nil}, - }, 0, 0) - return yylex -} - -func newLexer(in io.Reader) *lexer { - return newLexerWithInit(in, nil) -} - -// Text returns the matched text. -func (yylex *lexer) Text() string { - return yylex.stack[len(yylex.stack)-1].s -} - -// Line returns the current line number. -// The first line is 0. -func (yylex *lexer) Line() int { - if len(yylex.stack) == 0 { - return 0 - } - return yylex.stack[len(yylex.stack)-1].line -} - -// Column returns the current column number. -// The first column is 0. -func (yylex *lexer) Column() int { - if len(yylex.stack) == 0 { - return 0 - } - return yylex.stack[len(yylex.stack)-1].column -} - -func (yylex *lexer) next(lvl int) int { - if lvl == len(yylex.stack) { - l, c := 0, 0 - if lvl > 0 { - l, c = yylex.stack[lvl-1].line, yylex.stack[lvl-1].column - } - yylex.stack = append(yylex.stack, frame{0, "", l, c}) - } - if lvl == len(yylex.stack)-1 { - p := &yylex.stack[lvl] - *p = <-yylex.ch - yylex.stale = false - } else { - yylex.stale = true - } - return yylex.stack[lvl].i -} -func (yylex *lexer) pop() { - yylex.stack = yylex.stack[:len(yylex.stack)-1] -} -func (yylex lexer) Error(e string) { - panic(e) -} - -// Lex runs the lexer. Always returns 0. -// When the -s option is given, this function is not generated; -// instead, the NN_FUN macro runs the lexer. -func (yylex *lexer) Lex(lval *yySymType) int { -OUTER0: - for { - switch yylex.next(0) { - case 0: - { - lval.s = yylex.Text()[1 : len(yylex.Text())-1] - logDebugTokens("PHRASE - %s", lval.s) - return tPHRASE - } - case 1: - { - lval.s = yylex.Text()[1 : len(yylex.Text())-1] - logDebugTokens("REGEXP - %s", lval.s) - return tREGEXP - } - case 2: - { - logDebugTokens("PLUS") - return tPLUS - } - case 3: - { - logDebugTokens("MINUS") - return tMINUS - } - case 4: - { - logDebugTokens("COLON") - return tCOLON - } - case 5: - { - logDebugTokens("BOOST") - return tBOOST - } - case 6: - { - logDebugTokens("LPAREN") - return tLPAREN - } - case 7: - { - logDebugTokens("RPAREN") - return tRPAREN - } - case 8: - { - logDebugTokens("GREATER") - return tGREATER - } - case 9: - { - logDebugTokens("LESS") - return tLESS - } - case 10: - { - logDebugTokens("EQUAL") - return tEQUAL - } - case 11: - { - lval.s = yylex.Text()[1:] - logDebugTokens("TILDENUMBER - %s", lval.s) - return tTILDENUMBER - } - case 12: - { - logDebugTokens("TILDE") - return tTILDE - } - case 13: - { - lval.s = yylex.Text() - logDebugTokens("NUMBER - %s", lval.s) - return tNUMBER - } - case 14: - { - logDebugTokens("WHITESPACE (count=%d)", len(yylex.Text())) /* eat up whitespace */ - } - case 15: - { - lval.s = yylex.Text() - logDebugTokens("STRING - %s", lval.s) - return tSTRING - } - case 16: - { - lval.s = yylex.Text() - logDebugTokens("WILD - %s", lval.s) - return tWILD - } - default: - break OUTER0 - } - continue - } - yylex.pop() - - return 0 -} -func logDebugTokens(format string, v ...interface{}) { - if debugLexer { - logger.Printf(format, v...) - } -} diff --git a/vendor/github.com/blevesearch/bleve/query_string.y b/vendor/github.com/blevesearch/bleve/query_string.y index c262c8d..4849ac3 100644 --- a/vendor/github.com/blevesearch/bleve/query_string.y +++ b/vendor/github.com/blevesearch/bleve/query_string.y @@ -1,6 +1,10 @@ %{ package bleve -import "strconv" +import ( + "fmt" + "strconv" + "strings" +) func logDebugGrammar(format string, v ...interface{}) { if debugParser { @@ -15,20 +19,17 @@ n int f float64 q Query} -%token tSTRING tPHRASE tPLUS tMINUS tCOLON tBOOST tLPAREN tRPAREN tNUMBER tSTRING tGREATER tLESS -tEQUAL tTILDE tTILDENUMBER tREGEXP tWILD +%token tSTRING tPHRASE tPLUS tMINUS tCOLON tBOOST tNUMBER tSTRING tGREATER tLESS +tEQUAL tTILDE %type tSTRING -%type tWILD -%type tREGEXP %type tPHRASE %type tNUMBER -%type tTILDENUMBER +%type tTILDE +%type tBOOST %type searchBase %type searchSuffix %type searchPrefix -%type searchMustMustNot -%type searchBoost %% @@ -66,12 +67,6 @@ searchPrefix: $$ = queryShould } | -searchMustMustNot { - $$ = $1 -} -; - -searchMustMustNot: tPLUS { logDebugGrammar("PLUS") $$ = queryMust @@ -86,76 +81,39 @@ searchBase: tSTRING { str := $1 logDebugGrammar("STRING - %s", str) - q := NewMatchQuery(str) - $$ = q -} -| -tREGEXP { - str := $1 - logDebugGrammar("REGEXP - %s", str) - q := NewRegexpQuery(str) - $$ = q -} -| -tWILD { - str := $1 - logDebugGrammar("WILDCARD - %s", str) - q := NewWildcardQuery(str) + var q Query + if strings.HasPrefix(str, "/") && strings.HasSuffix(str, "/") { + q = NewRegexpQuery(str[1:len(str)-1]) + } else if strings.ContainsAny(str, "*?"){ + q = NewWildcardQuery(str) + } else { + q = NewMatchQuery(str) + } $$ = q } | tSTRING tTILDE { str := $1 - logDebugGrammar("FUZZY STRING - %s", str) + fuzziness, err := strconv.ParseFloat($2, 64) + if err != nil { + yylex.(*lexerWrapper).lex.Error(fmt.Sprintf("invalid fuzziness value: %v", err)) + } + logDebugGrammar("FUZZY STRING - %s %f", str, fuzziness) q := NewMatchQuery(str) - q.SetFuzziness(1) + q.SetFuzziness(int(fuzziness)) $$ = q } | tSTRING tCOLON tSTRING tTILDE { field := $1 str := $3 - logDebugGrammar("FIELD - %s FUZZY STRING - %s", field, str) - q := NewMatchQuery(str) - q.SetFuzziness(1) - q.SetField(field) - $$ = q -} -| -tSTRING tTILDENUMBER { - str := $1 - fuzziness, _ := strconv.ParseFloat($2, 64) - logDebugGrammar("FUZZY STRING - %s", str) + fuzziness, err := strconv.ParseFloat($4, 64) + if err != nil { + yylex.(*lexerWrapper).lex.Error(fmt.Sprintf("invalid fuzziness value: %v", err)) + } + logDebugGrammar("FIELD - %s FUZZY STRING - %s %f", field, str, fuzziness) q := NewMatchQuery(str) q.SetFuzziness(int(fuzziness)) - $$ = q -} -| -tSTRING tCOLON tSTRING tTILDENUMBER { - field := $1 - str := $3 - fuzziness, _ := strconv.ParseFloat($4, 64) - logDebugGrammar("FIELD - %s FUZZY-%f STRING - %s", field, fuzziness, str) - q := NewMatchQuery(str) - q.SetFuzziness(int(fuzziness)) - q.SetField(field) - $$ = q -} -| -tSTRING tCOLON tREGEXP { - field := $1 - str := $3 - logDebugGrammar("FIELD - %s REGEXP - %s", field, str) - q := NewRegexpQuery(str) - q.SetField(field) - $$ = q -} -| -tSTRING tCOLON tWILD { - field := $1 - str := $3 - logDebugGrammar("FIELD - %s WILD - %s", field, str) - q := NewWildcardQuery(str) q.SetField(field) $$ = q } @@ -178,7 +136,15 @@ tSTRING tCOLON tSTRING { field := $1 str := $3 logDebugGrammar("FIELD - %s STRING - %s", field, str) - q := NewMatchQuery(str).SetField(field) + var q Query + if strings.HasPrefix(str, "/") && strings.HasSuffix(str, "/") { + q = NewRegexpQuery(str[1:len(str)-1]) + } else if strings.ContainsAny(str, "*?"){ + q = NewWildcardQuery(str) + } else { + q = NewMatchQuery(str) + } + q.SetField(field) $$ = q } | @@ -232,13 +198,46 @@ tSTRING tCOLON tLESS tEQUAL tNUMBER { logDebugGrammar("FIELD - LESS THAN OR EQUAL %f", max) q := NewNumericRangeInclusiveQuery(nil, &max, nil, &maxInclusive).SetField(field) $$ = q -}; +} +| +tSTRING tCOLON tGREATER tPHRASE { + field := $1 + minInclusive := false + phrase := $4 -searchBoost: -tBOOST tNUMBER { - boost, _ := strconv.ParseFloat($2, 64) - $$ = boost - logDebugGrammar("BOOST %f", boost) + logDebugGrammar("FIELD - GREATER THAN DATE %s", phrase) + q := NewDateRangeInclusiveQuery(&phrase, nil, &minInclusive, nil).SetField(field) + $$ = q +} +| +tSTRING tCOLON tGREATER tEQUAL tPHRASE { + field := $1 + minInclusive := true + phrase := $5 + + logDebugGrammar("FIELD - GREATER THAN OR EQUAL DATE %s", phrase) + q := NewDateRangeInclusiveQuery(&phrase, nil, &minInclusive, nil).SetField(field) + $$ = q +} +| +tSTRING tCOLON tLESS tPHRASE { + field := $1 + maxInclusive := false + phrase := $4 + + logDebugGrammar("FIELD - LESS THAN DATE %s", phrase) + q := NewDateRangeInclusiveQuery(nil, &phrase, nil, &maxInclusive).SetField(field) + $$ = q +} +| +tSTRING tCOLON tLESS tEQUAL tPHRASE { + field := $1 + maxInclusive := true + phrase := $5 + + logDebugGrammar("FIELD - LESS THAN OR EQUAL DATE %s", phrase) + q := NewDateRangeInclusiveQuery(nil, &phrase, nil, &maxInclusive).SetField(field) + $$ = q }; searchSuffix: @@ -246,6 +245,11 @@ searchSuffix: $$ = 1.0 } | -searchBoost { - +tBOOST { + boost, err := strconv.ParseFloat($1, 64) + if err != nil { + yylex.(*lexerWrapper).lex.Error(fmt.Sprintf("invalid boost value: %v", err)) + } + $$ = boost + logDebugGrammar("BOOST %f", boost) }; diff --git a/vendor/github.com/blevesearch/bleve/query_string.y.go b/vendor/github.com/blevesearch/bleve/query_string.y.go index 31cd371..df75cd0 100644 --- a/vendor/github.com/blevesearch/bleve/query_string.y.go +++ b/vendor/github.com/blevesearch/bleve/query_string.y.go @@ -3,7 +3,11 @@ package bleve import __yyfmt__ "fmt" //line query_string.y:2 -import "strconv" +import ( + "fmt" + "strconv" + "strings" +) func logDebugGrammar(format string, v ...interface{}) { if debugParser { @@ -11,7 +15,7 @@ func logDebugGrammar(format string, v ...interface{}) { } } -//line query_string.y:12 +//line query_string.y:16 type yySymType struct { yys int s string @@ -26,16 +30,11 @@ const tPLUS = 57348 const tMINUS = 57349 const tCOLON = 57350 const tBOOST = 57351 -const tLPAREN = 57352 -const tRPAREN = 57353 -const tNUMBER = 57354 -const tGREATER = 57355 -const tLESS = 57356 -const tEQUAL = 57357 -const tTILDE = 57358 -const tTILDENUMBER = 57359 -const tREGEXP = 57360 -const tWILD = 57361 +const tNUMBER = 57352 +const tGREATER = 57353 +const tLESS = 57354 +const tEQUAL = 57355 +const tTILDE = 57356 var yyToknames = [...]string{ "$end", @@ -47,16 +46,11 @@ var yyToknames = [...]string{ "tMINUS", "tCOLON", "tBOOST", - "tLPAREN", - "tRPAREN", "tNUMBER", "tGREATER", "tLESS", "tEQUAL", "tTILDE", - "tTILDENUMBER", - "tREGEXP", - "tWILD", } var yyStatenames = [...]string{} @@ -74,57 +68,57 @@ var yyExca = [...]int{ -2, 5, } -const yyNprod = 30 +const yyNprod = 26 const yyPrivate = 57344 var yyTokenNames []string var yyStates []string -const yyLast = 36 +const yyLast = 31 var yyAct = [...]int{ - 22, 26, 36, 10, 14, 29, 30, 35, 25, 27, - 28, 13, 19, 33, 23, 24, 34, 11, 12, 31, - 18, 20, 32, 21, 17, 6, 7, 2, 3, 1, - 16, 8, 5, 4, 15, 9, + 16, 18, 21, 13, 27, 24, 17, 19, 20, 25, + 22, 15, 26, 23, 9, 11, 31, 14, 29, 3, + 10, 30, 2, 28, 5, 6, 7, 1, 4, 12, + 8, } var yyPact = [...]int{ - 19, -1000, -1000, 19, -1, -1000, -1000, -1000, -1000, 15, - 4, -1000, -1000, -1000, -1000, -1000, -1000, 11, -1000, -4, - -1000, -1000, -11, -1000, -1000, -1000, -1000, 7, 1, -1000, - -1000, -1000, -5, -1000, -10, -1000, -1000, + 18, -1000, -1000, 18, 10, -1000, -1000, -1000, -6, 3, + -1000, -1000, -1000, -1000, -1000, -4, -12, -1000, -1000, 0, + -1, -1000, -1000, 13, -1000, -1000, 11, -1000, -1000, -1000, + -1000, -1000, } var yyPgo = [...]int{ - 0, 35, 34, 33, 32, 30, 29, 27, 28, + 0, 30, 29, 28, 27, 22, 19, } var yyR1 = [...]int{ - 0, 6, 7, 7, 8, 3, 3, 4, 4, 1, + 0, 4, 5, 5, 6, 3, 3, 3, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, - 1, 1, 1, 1, 1, 1, 1, 5, 2, 2, + 1, 1, 1, 1, 2, 2, } var yyR2 = [...]int{ - 0, 1, 2, 1, 3, 0, 1, 1, 1, 1, - 1, 1, 2, 4, 2, 4, 3, 3, 1, 1, - 3, 3, 3, 4, 5, 4, 5, 2, 0, 1, + 0, 1, 2, 1, 3, 0, 1, 1, 1, 2, + 4, 1, 1, 3, 3, 3, 4, 5, 4, 5, + 4, 5, 4, 5, 0, 1, } var yyChk = [...]int{ - -1000, -6, -7, -8, -3, -4, 6, 7, -7, -1, - 4, 18, 19, 12, 5, -2, -5, 9, 16, 8, - 17, 12, 4, 18, 19, 12, 5, 13, 14, 16, - 17, 12, 15, 12, 15, 12, 12, + -1000, -4, -5, -6, -3, 6, 7, -5, -1, 4, + 10, 5, -2, 9, 14, 8, 4, 10, 5, 11, + 12, 14, 10, 13, 5, 10, 13, 5, 10, 5, + 10, 5, } var yyDef = [...]int{ - 5, -2, 1, -2, 0, 6, 7, 8, 2, 28, - 9, 10, 11, 18, 19, 4, 29, 0, 12, 0, - 14, 27, 20, 16, 17, 21, 22, 0, 0, 13, - 15, 23, 0, 25, 0, 24, 26, + 5, -2, 1, -2, 0, 6, 7, 2, 24, 8, + 11, 12, 4, 25, 9, 0, 13, 14, 15, 0, + 0, 10, 16, 0, 20, 18, 0, 22, 17, 21, + 19, 23, } var yyTok1 = [...]int{ @@ -133,7 +127,7 @@ var yyTok1 = [...]int{ var yyTok2 = [...]int{ 2, 3, 4, 5, 6, 7, 8, 9, 10, 11, - 12, 13, 14, 15, 16, 17, 18, 19, + 12, 13, 14, } var yyTok3 = [...]int{ 0, @@ -478,25 +472,25 @@ yydefault: case 1: yyDollar = yyS[yypt-1 : yypt+1] - //line query_string.y:36 + //line query_string.y:37 { logDebugGrammar("INPUT") } case 2: yyDollar = yyS[yypt-2 : yypt+1] - //line query_string.y:41 + //line query_string.y:42 { logDebugGrammar("SEARCH PARTS") } case 3: yyDollar = yyS[yypt-1 : yypt+1] - //line query_string.y:45 + //line query_string.y:46 { logDebugGrammar("SEARCH PART") } case 4: yyDollar = yyS[yypt-3 : yypt+1] - //line query_string.y:50 + //line query_string.y:51 { query := yyDollar[2].q query.SetBoost(yyDollar[3].f) @@ -511,146 +505,109 @@ yydefault: } case 5: yyDollar = yyS[yypt-0 : yypt+1] - //line query_string.y:65 + //line query_string.y:66 { yyVAL.n = queryShould } case 6: yyDollar = yyS[yypt-1 : yypt+1] - //line query_string.y:69 + //line query_string.y:70 { - yyVAL.n = yyDollar[1].n + logDebugGrammar("PLUS") + yyVAL.n = queryMust } case 7: yyDollar = yyS[yypt-1 : yypt+1] //line query_string.y:75 - { - logDebugGrammar("PLUS") - yyVAL.n = queryMust - } - case 8: - yyDollar = yyS[yypt-1 : yypt+1] - //line query_string.y:80 { logDebugGrammar("MINUS") yyVAL.n = queryMustNot } - case 9: + case 8: yyDollar = yyS[yypt-1 : yypt+1] - //line query_string.y:86 + //line query_string.y:81 { str := yyDollar[1].s logDebugGrammar("STRING - %s", str) + var q Query + if strings.HasPrefix(str, "/") && strings.HasSuffix(str, "/") { + q = NewRegexpQuery(str[1 : len(str)-1]) + } else if strings.ContainsAny(str, "*?") { + q = NewWildcardQuery(str) + } else { + q = NewMatchQuery(str) + } + yyVAL.q = q + } + case 9: + yyDollar = yyS[yypt-2 : yypt+1] + //line query_string.y:95 + { + str := yyDollar[1].s + fuzziness, err := strconv.ParseFloat(yyDollar[2].s, 64) + if err != nil { + yylex.(*lexerWrapper).lex.Error(fmt.Sprintf("invalid fuzziness value: %v", err)) + } + logDebugGrammar("FUZZY STRING - %s %f", str, fuzziness) q := NewMatchQuery(str) + q.SetFuzziness(int(fuzziness)) yyVAL.q = q } case 10: - yyDollar = yyS[yypt-1 : yypt+1] - //line query_string.y:93 + yyDollar = yyS[yypt-4 : yypt+1] + //line query_string.y:107 { - str := yyDollar[1].s - logDebugGrammar("REGEXP - %s", str) - q := NewRegexpQuery(str) + field := yyDollar[1].s + str := yyDollar[3].s + fuzziness, err := strconv.ParseFloat(yyDollar[4].s, 64) + if err != nil { + yylex.(*lexerWrapper).lex.Error(fmt.Sprintf("invalid fuzziness value: %v", err)) + } + logDebugGrammar("FIELD - %s FUZZY STRING - %s %f", field, str, fuzziness) + q := NewMatchQuery(str) + q.SetFuzziness(int(fuzziness)) + q.SetField(field) yyVAL.q = q } case 11: yyDollar = yyS[yypt-1 : yypt+1] - //line query_string.y:100 - { - str := yyDollar[1].s - logDebugGrammar("WILDCARD - %s", str) - q := NewWildcardQuery(str) - yyVAL.q = q - } - case 12: - yyDollar = yyS[yypt-2 : yypt+1] - //line query_string.y:107 - { - str := yyDollar[1].s - logDebugGrammar("FUZZY STRING - %s", str) - q := NewMatchQuery(str) - q.SetFuzziness(1) - yyVAL.q = q - } - case 13: - yyDollar = yyS[yypt-4 : yypt+1] - //line query_string.y:115 - { - field := yyDollar[1].s - str := yyDollar[3].s - logDebugGrammar("FIELD - %s FUZZY STRING - %s", field, str) - q := NewMatchQuery(str) - q.SetFuzziness(1) - q.SetField(field) - yyVAL.q = q - } - case 14: - yyDollar = yyS[yypt-2 : yypt+1] - //line query_string.y:125 - { - str := yyDollar[1].s - fuzziness, _ := strconv.ParseFloat(yyDollar[2].s, 64) - logDebugGrammar("FUZZY STRING - %s", str) - q := NewMatchQuery(str) - q.SetFuzziness(int(fuzziness)) - yyVAL.q = q - } - case 15: - yyDollar = yyS[yypt-4 : yypt+1] - //line query_string.y:134 - { - field := yyDollar[1].s - str := yyDollar[3].s - fuzziness, _ := strconv.ParseFloat(yyDollar[4].s, 64) - logDebugGrammar("FIELD - %s FUZZY-%f STRING - %s", field, fuzziness, str) - q := NewMatchQuery(str) - q.SetFuzziness(int(fuzziness)) - q.SetField(field) - yyVAL.q = q - } - case 16: - yyDollar = yyS[yypt-3 : yypt+1] - //line query_string.y:145 - { - field := yyDollar[1].s - str := yyDollar[3].s - logDebugGrammar("FIELD - %s REGEXP - %s", field, str) - q := NewRegexpQuery(str) - q.SetField(field) - yyVAL.q = q - } - case 17: - yyDollar = yyS[yypt-3 : yypt+1] - //line query_string.y:154 - { - field := yyDollar[1].s - str := yyDollar[3].s - logDebugGrammar("FIELD - %s WILD - %s", field, str) - q := NewWildcardQuery(str) - q.SetField(field) - yyVAL.q = q - } - case 18: - yyDollar = yyS[yypt-1 : yypt+1] - //line query_string.y:163 + //line query_string.y:121 { str := yyDollar[1].s logDebugGrammar("STRING - %s", str) q := NewMatchQuery(str) yyVAL.q = q } - case 19: + case 12: yyDollar = yyS[yypt-1 : yypt+1] - //line query_string.y:170 + //line query_string.y:128 { phrase := yyDollar[1].s logDebugGrammar("PHRASE - %s", phrase) q := NewMatchPhraseQuery(phrase) yyVAL.q = q } - case 20: + case 13: yyDollar = yyS[yypt-3 : yypt+1] - //line query_string.y:177 + //line query_string.y:135 + { + field := yyDollar[1].s + str := yyDollar[3].s + logDebugGrammar("FIELD - %s STRING - %s", field, str) + var q Query + if strings.HasPrefix(str, "/") && strings.HasSuffix(str, "/") { + q = NewRegexpQuery(str[1 : len(str)-1]) + } else if strings.ContainsAny(str, "*?") { + q = NewWildcardQuery(str) + } else { + q = NewMatchQuery(str) + } + q.SetField(field) + yyVAL.q = q + } + case 14: + yyDollar = yyS[yypt-3 : yypt+1] + //line query_string.y:151 { field := yyDollar[1].s str := yyDollar[3].s @@ -658,19 +615,9 @@ yydefault: q := NewMatchQuery(str).SetField(field) yyVAL.q = q } - case 21: + case 15: yyDollar = yyS[yypt-3 : yypt+1] - //line query_string.y:185 - { - field := yyDollar[1].s - str := yyDollar[3].s - logDebugGrammar("FIELD - %s STRING - %s", field, str) - q := NewMatchQuery(str).SetField(field) - yyVAL.q = q - } - case 22: - yyDollar = yyS[yypt-3 : yypt+1] - //line query_string.y:193 + //line query_string.y:159 { field := yyDollar[1].s phrase := yyDollar[3].s @@ -678,9 +625,9 @@ yydefault: q := NewMatchPhraseQuery(phrase).SetField(field) yyVAL.q = q } - case 23: + case 16: yyDollar = yyS[yypt-4 : yypt+1] - //line query_string.y:201 + //line query_string.y:167 { field := yyDollar[1].s min, _ := strconv.ParseFloat(yyDollar[4].s, 64) @@ -689,9 +636,9 @@ yydefault: q := NewNumericRangeInclusiveQuery(&min, nil, &minInclusive, nil).SetField(field) yyVAL.q = q } - case 24: + case 17: yyDollar = yyS[yypt-5 : yypt+1] - //line query_string.y:210 + //line query_string.y:176 { field := yyDollar[1].s min, _ := strconv.ParseFloat(yyDollar[5].s, 64) @@ -700,9 +647,9 @@ yydefault: q := NewNumericRangeInclusiveQuery(&min, nil, &minInclusive, nil).SetField(field) yyVAL.q = q } - case 25: + case 18: yyDollar = yyS[yypt-4 : yypt+1] - //line query_string.y:219 + //line query_string.y:185 { field := yyDollar[1].s max, _ := strconv.ParseFloat(yyDollar[4].s, 64) @@ -711,9 +658,9 @@ yydefault: q := NewNumericRangeInclusiveQuery(nil, &max, nil, &maxInclusive).SetField(field) yyVAL.q = q } - case 26: + case 19: yyDollar = yyS[yypt-5 : yypt+1] - //line query_string.y:228 + //line query_string.y:194 { field := yyDollar[1].s max, _ := strconv.ParseFloat(yyDollar[5].s, 64) @@ -722,25 +669,70 @@ yydefault: q := NewNumericRangeInclusiveQuery(nil, &max, nil, &maxInclusive).SetField(field) yyVAL.q = q } - case 27: - yyDollar = yyS[yypt-2 : yypt+1] - //line query_string.y:238 + case 20: + yyDollar = yyS[yypt-4 : yypt+1] + //line query_string.y:203 { - boost, _ := strconv.ParseFloat(yyDollar[2].s, 64) - yyVAL.f = boost - logDebugGrammar("BOOST %f", boost) + field := yyDollar[1].s + minInclusive := false + phrase := yyDollar[4].s + + logDebugGrammar("FIELD - GREATER THAN DATE %s", phrase) + q := NewDateRangeInclusiveQuery(&phrase, nil, &minInclusive, nil).SetField(field) + yyVAL.q = q } - case 28: + case 21: + yyDollar = yyS[yypt-5 : yypt+1] + //line query_string.y:213 + { + field := yyDollar[1].s + minInclusive := true + phrase := yyDollar[5].s + + logDebugGrammar("FIELD - GREATER THAN OR EQUAL DATE %s", phrase) + q := NewDateRangeInclusiveQuery(&phrase, nil, &minInclusive, nil).SetField(field) + yyVAL.q = q + } + case 22: + yyDollar = yyS[yypt-4 : yypt+1] + //line query_string.y:223 + { + field := yyDollar[1].s + maxInclusive := false + phrase := yyDollar[4].s + + logDebugGrammar("FIELD - LESS THAN DATE %s", phrase) + q := NewDateRangeInclusiveQuery(nil, &phrase, nil, &maxInclusive).SetField(field) + yyVAL.q = q + } + case 23: + yyDollar = yyS[yypt-5 : yypt+1] + //line query_string.y:233 + { + field := yyDollar[1].s + maxInclusive := true + phrase := yyDollar[5].s + + logDebugGrammar("FIELD - LESS THAN OR EQUAL DATE %s", phrase) + q := NewDateRangeInclusiveQuery(nil, &phrase, nil, &maxInclusive).SetField(field) + yyVAL.q = q + } + case 24: yyDollar = yyS[yypt-0 : yypt+1] - //line query_string.y:245 + //line query_string.y:244 { yyVAL.f = 1.0 } - case 29: + case 25: yyDollar = yyS[yypt-1 : yypt+1] - //line query_string.y:249 + //line query_string.y:248 { - + boost, err := strconv.ParseFloat(yyDollar[1].s, 64) + if err != nil { + yylex.(*lexerWrapper).lex.Error(fmt.Sprintf("invalid boost value: %v", err)) + } + yyVAL.f = boost + logDebugGrammar("BOOST %f", boost) } } goto yystack /* stack new state and value */ diff --git a/vendor/github.com/blevesearch/bleve/query_string_lex.go b/vendor/github.com/blevesearch/bleve/query_string_lex.go new file mode 100644 index 0000000..8bd4a01 --- /dev/null +++ b/vendor/github.com/blevesearch/bleve/query_string_lex.go @@ -0,0 +1,317 @@ +// Copyright (c) 2016 Couchbase, Inc. +// Licensed under the Apache License, Version 2.0 (the "License"); you may not use this file +// except in compliance with the License. You may obtain a copy of the License at +// http://www.apache.org/licenses/LICENSE-2.0 +// Unless required by applicable law or agreed to in writing, software distributed under the +// License is distributed on an "AS IS" BASIS, WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, +// either express or implied. See the License for the specific language governing permissions +// and limitations under the License. + +package bleve + +import ( + "bufio" + "io" + "strings" + "unicode" +) + +const reservedChars = "+-=&|>', '<', '=': + l.buf += string(next) + return singleCharOpState, true + case '^': + return inBoostState, true + case '~': + return inTildeState, true + } + + switch { + case !l.inEscape && next == '\\': + l.inEscape = true + return startState, true + case unicode.IsDigit(next): + l.buf += string(next) + return inNumOrStrState, true + case !unicode.IsSpace(next): + l.buf += string(next) + return inStrState, true + } + + // doesnt look like anything, just eat it and stay here + l.reset() + return startState, true +} + +func inPhraseState(l *queryStringLex, next rune, eof bool) (lexState, bool) { + // unterminated phrase eats the phrase + if eof { + l.Error("unterminated quote") + return nil, false + } + + // only a non-escaped " ends the phrase + if !l.inEscape && next == '"' { + // end phrase + l.nextTokenType = tPHRASE + l.nextToken = &yySymType{ + s: l.buf, + } + logDebugTokens("PHRASE - '%s'", l.nextToken.s) + l.reset() + return startState, true + } else if !l.inEscape && next == '\\' { + l.inEscape = true + } else if l.inEscape { + // if in escape, end it + l.inEscape = false + l.buf += unescape(string(next)) + } else { + l.buf += string(next) + } + + return inPhraseState, true +} + +func singleCharOpState(l *queryStringLex, next rune, eof bool) (lexState, bool) { + l.nextToken = &yySymType{} + + switch l.buf { + case "+": + l.nextTokenType = tPLUS + logDebugTokens("PLUS") + case "-": + l.nextTokenType = tMINUS + logDebugTokens("MINUS") + case ":": + l.nextTokenType = tCOLON + logDebugTokens("COLON") + case ">": + l.nextTokenType = tGREATER + logDebugTokens("GREATER") + case "<": + l.nextTokenType = tLESS + logDebugTokens("LESS") + case "=": + l.nextTokenType = tEQUAL + logDebugTokens("EQUAL") + } + + l.reset() + return startState, false +} + +func inBoostState(l *queryStringLex, next rune, eof bool) (lexState, bool) { + + // only a non-escaped space ends the boost (or eof) + if eof || (!l.inEscape && next == ' ') { + // end boost + l.nextTokenType = tBOOST + if l.buf == "" { + l.buf = "1" + } + l.nextToken = &yySymType{ + s: l.buf, + } + logDebugTokens("BOOST - '%s'", l.nextToken.s) + l.reset() + return startState, true + } else if !l.inEscape && next == '\\' { + l.inEscape = true + } else if l.inEscape { + // if in escape, end it + l.inEscape = false + l.buf += unescape(string(next)) + } else { + l.buf += string(next) + } + + return inBoostState, true +} + +func inTildeState(l *queryStringLex, next rune, eof bool) (lexState, bool) { + + // only a non-escaped space ends the tilde (or eof) + if eof || (!l.inEscape && next == ' ') { + // end tilde + l.nextTokenType = tTILDE + if l.buf == "" { + l.buf = "1" + } + l.nextToken = &yySymType{ + s: l.buf, + } + logDebugTokens("TILDE - '%s'", l.nextToken.s) + l.reset() + return startState, true + } else if !l.inEscape && next == '\\' { + l.inEscape = true + } else if l.inEscape { + // if in escape, end it + l.inEscape = false + l.buf += unescape(string(next)) + } else { + l.buf += string(next) + } + + return inTildeState, true +} + +func inNumOrStrState(l *queryStringLex, next rune, eof bool) (lexState, bool) { + // only a non-escaped space ends the tilde (or eof) + if eof || (!l.inEscape && next == ' ') { + // end number + l.nextTokenType = tNUMBER + l.nextToken = &yySymType{ + s: l.buf, + } + logDebugTokens("NUMBER - '%s'", l.nextToken.s) + l.reset() + return startState, true + } else if !l.inEscape && next == '\\' { + l.inEscape = true + return inNumOrStrState, true + } else if l.inEscape { + // if in escape, end it + l.inEscape = false + l.buf += unescape(string(next)) + // go directly to string, no successfully or unsuccessfully + // escaped string results in a valid number + return inStrState, true + } + + // see where to go + if !l.seenDot && next == '.' { + // stay in this state + l.buf += string(next) + return inNumOrStrState, true + } else if unicode.IsDigit(next) { + l.buf += string(next) + return inNumOrStrState, true + } + + // doesn't look like an number, transition + l.buf += string(next) + return inStrState, true +} + +func inStrState(l *queryStringLex, next rune, eof bool) (lexState, bool) { + // end on non-escped space, colon, tilde, boost (or eof) + if eof || (!l.inEscape && (next == ' ' || next == ':' || next == '^' || next == '~')) { + // end string + l.nextTokenType = tSTRING + l.nextToken = &yySymType{ + s: l.buf, + } + logDebugTokens("STRING - '%s'", l.nextToken.s) + l.reset() + + consumed := true + if !eof && (next == ':' || next == '^' || next == '~') { + consumed = false + } + + return startState, consumed + } else if !l.inEscape && next == '\\' { + l.inEscape = true + } else if l.inEscape { + // if in escape, end it + l.inEscape = false + l.buf += unescape(string(next)) + } else { + l.buf += string(next) + } + + return inStrState, true +} + +func logDebugTokens(format string, v ...interface{}) { + if debugLexer { + logger.Printf(format, v...) + } +} diff --git a/vendor/github.com/blevesearch/bleve/query_string_parser.go b/vendor/github.com/blevesearch/bleve/query_string_parser.go index 3b03552..f9875df 100644 --- a/vendor/github.com/blevesearch/bleve/query_string_parser.go +++ b/vendor/github.com/blevesearch/bleve/query_string_parser.go @@ -7,13 +7,13 @@ // either express or implied. See the License for the specific language governing permissions // and limitations under the License. -//go:generate nex query_string.nex -//go:generate sed -i "" -e s/Lexer/lexer/g query_string.nn.go -//go:generate sed -i "" -e s/Newlexer/newLexer/g query_string.nn.go -//go:generate sed -i "" -e s/debuglexer/debugLexer/g query_string.nn.go -//go:generate go fmt query_string.nn.go //go:generate go tool yacc -o query_string.y.go query_string.y -//go:generate sed -i "" -e 1d query_string.y.go +//go:generate sed -i.tmp -e 1d query_string.y.go +//go:generate rm query_string.y.go.tmp + +// note: OSX sed and gnu sed handle the -i (in-place) option differently. +// using -i.tmp works on both, at the expense of having to remove +// the unsightly .tmp files package bleve @@ -25,22 +25,21 @@ import ( var debugParser bool var debugLexer bool -func parseQuerySyntax(query string, mapping *IndexMapping) (rq Query, err error) { - lex := newLexerWrapper(newLexer(strings.NewReader(query))) +func parseQuerySyntax(query string) (rq Query, err error) { + lex := newLexerWrapper(newQueryStringLex(strings.NewReader(query))) doParse(lex) if len(lex.errs) > 0 { return nil, fmt.Errorf(strings.Join(lex.errs, "\n")) - } else { - return lex.query, nil } + return lex.query, nil } func doParse(lex *lexerWrapper) { defer func() { r := recover() if r != nil { - lex.Error("Errors while parsing.") + lex.errs = append(lex.errs, fmt.Sprintf("parse error: %v", r)) } }() @@ -54,23 +53,22 @@ const ( ) type lexerWrapper struct { - nex yyLexer + lex yyLexer errs []string query *booleanQuery } -func newLexerWrapper(nex yyLexer) *lexerWrapper { +func newLexerWrapper(lex yyLexer) *lexerWrapper { return &lexerWrapper{ - nex: nex, - errs: []string{}, + lex: lex, query: NewBooleanQuery(nil, nil, nil), } } -func (this *lexerWrapper) Lex(lval *yySymType) int { - return this.nex.Lex(lval) +func (l *lexerWrapper) Lex(lval *yySymType) int { + return l.lex.Lex(lval) } -func (this *lexerWrapper) Error(s string) { - this.errs = append(this.errs, s) +func (l *lexerWrapper) Error(s string) { + l.errs = append(l.errs, s) } diff --git a/vendor/github.com/blevesearch/bleve/registry/byte_array_converter.go b/vendor/github.com/blevesearch/bleve/registry/byte_array_converter.go deleted file mode 100644 index 0adb3cd..0000000 --- a/vendor/github.com/blevesearch/bleve/registry/byte_array_converter.go +++ /dev/null @@ -1,47 +0,0 @@ -// Copyright (c) 2014 Couchbase, Inc. -// Licensed under the Apache License, Version 2.0 (the "License"); you may not use this file -// except in compliance with the License. You may obtain a copy of the License at -// http://www.apache.org/licenses/LICENSE-2.0 -// Unless required by applicable law or agreed to in writing, software distributed under the -// License is distributed on an "AS IS" BASIS, WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, -// either express or implied. See the License for the specific language governing permissions -// and limitations under the License. - -package registry - -import ( - "fmt" - - "github.com/blevesearch/bleve/analysis" -) - -func RegisterByteArrayConverter(name string, constructor ByteArrayConverterConstructor) { - _, exists := byteArrayConverters[name] - if exists { - panic(fmt.Errorf("attempted to register duplicate byte array converter named '%s'", name)) - } - byteArrayConverters[name] = constructor -} - -type ByteArrayConverterConstructor func(config map[string]interface{}, cache *Cache) (analysis.ByteArrayConverter, error) -type ByteArrayConverterRegistry map[string]ByteArrayConverterConstructor - -func ByteArrayConverterByName(name string) ByteArrayConverterConstructor { - return byteArrayConverters[name] -} - -func ByteArrayConverterTypesAndInstances() ([]string, []string) { - emptyConfig := map[string]interface{}{} - emptyCache := NewCache() - types := make([]string, 0) - instances := make([]string, 0) - for name, cons := range byteArrayConverters { - _, err := cons(emptyConfig, emptyCache) - if err == nil { - instances = append(instances, name) - } else { - types = append(types, name) - } - } - return types, instances -} diff --git a/vendor/github.com/blevesearch/bleve/registry/registry.go b/vendor/github.com/blevesearch/bleve/registry/registry.go index 0d93ee6..1e5b061 100644 --- a/vendor/github.com/blevesearch/bleve/registry/registry.go +++ b/vendor/github.com/blevesearch/bleve/registry/registry.go @@ -19,8 +19,6 @@ import ( var stores = make(KVStoreRegistry, 0) var index_types = make(IndexTypeRegistry, 0) -var byteArrayConverters = make(ByteArrayConverterRegistry, 0) - // highlight var fragmentFormatters = make(FragmentFormatterRegistry, 0) var fragmenters = make(FragmenterRegistry, 0) diff --git a/vendor/github.com/blevesearch/bleve/search.go b/vendor/github.com/blevesearch/bleve/search.go index 380affd..dc6bc26 100644 --- a/vendor/github.com/blevesearch/bleve/search.go +++ b/vendor/github.com/blevesearch/bleve/search.go @@ -80,6 +80,30 @@ type FacetRequest struct { DateTimeRanges []*dateTimeRange `json:"date_ranges,omitempty"` } +func (fr *FacetRequest) Validate() error { + if len(fr.NumericRanges) > 0 && len(fr.DateTimeRanges) > 0 { + return fmt.Errorf("facet can only conain numeric ranges or date ranges, not both") + } + + nrNames := map[string]interface{}{} + for _, nr := range fr.NumericRanges { + if _, ok := nrNames[nr.Name]; ok { + return fmt.Errorf("numeric ranges contains duplicate name '%s'", nr.Name) + } + nrNames[nr.Name] = struct{}{} + } + + drNames := map[string]interface{}{} + for _, dr := range fr.DateTimeRanges { + if _, ok := drNames[dr.Name]; ok { + return fmt.Errorf("date ranges contains duplicate name '%s'", dr.Name) + } + drNames[dr.Name] = struct{}{} + } + + return nil +} + // NewFacetRequest creates a facet on the specified // field that limits the number of entries to the // specified size. @@ -116,6 +140,16 @@ func (fr *FacetRequest) AddNumericRange(name string, min, max *float64) { // FacetRequest objects for a single query. type FacetsRequest map[string]*FacetRequest +func (fr FacetsRequest) Validate() error { + for _, v := range fr { + err := v.Validate() + if err != nil { + return err + } + } + return nil +} + // HighlightRequest describes how field matches // should be highlighted. type HighlightRequest struct { @@ -157,6 +191,7 @@ func (h *HighlightRequest) AddField(field string) { // Facets describe the set of facets to be computed. // Explain triggers inclusion of additional search // result score explanations. +// Sort describes the desired order for the results to be returned. // // A special field named "*" can be used to return all fields. type SearchRequest struct { @@ -167,6 +202,16 @@ type SearchRequest struct { Fields []string `json:"fields"` Facets FacetsRequest `json:"facets"` Explain bool `json:"explain"` + Sort search.SortOrder `json:"sort"` +} + +func (sr *SearchRequest) Validate() error { + err := sr.Query.Validate() + if err != nil { + return err + } + + return sr.Facets.Validate() } // AddFacet adds a FacetRequest to this SearchRequest @@ -177,6 +222,21 @@ func (r *SearchRequest) AddFacet(facetName string, f *FacetRequest) { r.Facets[facetName] = f } +// SortBy changes the request to use the requested sort order +// this form uses the simplified syntax with an array of strings +// each string can either be a field name +// or the magic value _id and _score which refer to the doc id and search score +// any of these values can optionally be prefixed with - to reverse the order +func (r *SearchRequest) SortBy(order []string) { + so := search.ParseSortOrderStrings(order) + r.Sort = so +} + +// SortByCustom changes the request to use the requested sort order +func (r *SearchRequest) SortByCustom(order search.SortOrder) { + r.Sort = order +} + // UnmarshalJSON deserializes a JSON representation of // a SearchRequest func (r *SearchRequest) UnmarshalJSON(input []byte) error { @@ -188,6 +248,7 @@ func (r *SearchRequest) UnmarshalJSON(input []byte) error { Fields []string `json:"fields"` Facets FacetsRequest `json:"facets"` Explain bool `json:"explain"` + Sort []json.RawMessage `json:"sort"` } err := json.Unmarshal(input, &temp) @@ -200,6 +261,14 @@ func (r *SearchRequest) UnmarshalJSON(input []byte) error { } else { r.Size = *temp.Size } + if temp.Sort == nil { + r.Sort = search.SortOrder{&search.SortScore{Desc: true}} + } else { + r.Sort, err = search.ParseSortOrderJSON(temp.Sort) + if err != nil { + return err + } + } r.From = temp.From r.Explain = temp.Explain r.Highlight = temp.Highlight @@ -231,12 +300,14 @@ func NewSearchRequest(q Query) *SearchRequest { // NewSearchRequestOptions creates a new SearchRequest // for the Query, with the requested size, from // and explanation search parameters. +// By default results are ordered by score, descending. func NewSearchRequestOptions(q Query, size, from int, explain bool) *SearchRequest { return &SearchRequest{ Query: q, Size: size, From: from, Explain: explain, + Sort: search.SortOrder{&search.SortScore{Desc: true}}, } } @@ -252,6 +323,18 @@ func (iem IndexErrMap) MarshalJSON() ([]byte, error) { return json.Marshal(tmp) } +func (iem IndexErrMap) UnmarshalJSON(data []byte) error { + var tmp map[string]string + err := json.Unmarshal(data, &tmp) + if err != nil { + return err + } + for k, v := range tmp { + iem[k] = fmt.Errorf("%s", v) + } + return nil +} + // SearchStatus is a secion in the SearchResult reporting how many // underlying indexes were queried, how many were successful/failed // and a map of any errors that were encountered diff --git a/vendor/github.com/blevesearch/bleve/search/collector.go b/vendor/github.com/blevesearch/bleve/search/collector.go index 773c8d5..a6d9148 100644 --- a/vendor/github.com/blevesearch/bleve/search/collector.go +++ b/vendor/github.com/blevesearch/bleve/search/collector.go @@ -12,11 +12,13 @@ package search import ( "time" + "github.com/blevesearch/bleve/index" + "golang.org/x/net/context" ) type Collector interface { - Collect(ctx context.Context, searcher Searcher) error + Collect(ctx context.Context, searcher Searcher, reader index.IndexReader) error Results() DocumentMatchCollection Total() uint64 MaxScore() float64 diff --git a/vendor/github.com/blevesearch/bleve/search/collectors/collector_top_score.go b/vendor/github.com/blevesearch/bleve/search/collectors/collector_top_score.go deleted file mode 100644 index 2f00d13..0000000 --- a/vendor/github.com/blevesearch/bleve/search/collectors/collector_top_score.go +++ /dev/null @@ -1,149 +0,0 @@ -// Copyright (c) 2014 Couchbase, Inc. -// Licensed under the Apache License, Version 2.0 (the "License"); you may not use this file -// except in compliance with the License. You may obtain a copy of the License at -// http://www.apache.org/licenses/LICENSE-2.0 -// Unless required by applicable law or agreed to in writing, software distributed under the -// License is distributed on an "AS IS" BASIS, WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, -// either express or implied. See the License for the specific language governing permissions -// and limitations under the License. - -package collectors - -import ( - "container/list" - "time" - - "golang.org/x/net/context" - - "github.com/blevesearch/bleve/search" -) - -type TopScoreCollector struct { - k int - skip int - results *list.List - took time.Duration - maxScore float64 - total uint64 - facetsBuilder *search.FacetsBuilder -} - -func NewTopScorerCollector(k int) *TopScoreCollector { - return &TopScoreCollector{ - k: k, - skip: 0, - results: list.New(), - } -} - -func NewTopScorerSkipCollector(k, skip int) *TopScoreCollector { - return &TopScoreCollector{ - k: k, - skip: skip, - results: list.New(), - } -} - -func (tksc *TopScoreCollector) Total() uint64 { - return tksc.total -} - -func (tksc *TopScoreCollector) MaxScore() float64 { - return tksc.maxScore -} - -func (tksc *TopScoreCollector) Took() time.Duration { - return tksc.took -} - -func (tksc *TopScoreCollector) Collect(ctx context.Context, searcher search.Searcher) error { - startTime := time.Now() - var err error - var next *search.DocumentMatch - select { - case <-ctx.Done(): - return ctx.Err() - default: - next, err = searcher.Next() - } - for err == nil && next != nil { - select { - case <-ctx.Done(): - return ctx.Err() - default: - tksc.collectSingle(next) - if tksc.facetsBuilder != nil { - err = tksc.facetsBuilder.Update(next) - if err != nil { - break - } - } - next, err = searcher.Next() - } - } - // compute search duration - tksc.took = time.Since(startTime) - if err != nil { - return err - } - return nil -} - -func (tksc *TopScoreCollector) collectSingle(dm *search.DocumentMatch) { - // increment total hits - tksc.total++ - - // update max score - if dm.Score > tksc.maxScore { - tksc.maxScore = dm.Score - } - - for e := tksc.results.Front(); e != nil; e = e.Next() { - curr := e.Value.(*search.DocumentMatch) - if dm.Score < curr.Score { - - tksc.results.InsertBefore(dm, e) - // if we just made the list too long - if tksc.results.Len() > (tksc.k + tksc.skip) { - // remove the head - tksc.results.Remove(tksc.results.Front()) - } - return - } - } - // if we got to the end, we still have to add it - tksc.results.PushBack(dm) - if tksc.results.Len() > (tksc.k + tksc.skip) { - // remove the head - tksc.results.Remove(tksc.results.Front()) - } -} - -func (tksc *TopScoreCollector) Results() search.DocumentMatchCollection { - if tksc.results.Len()-tksc.skip > 0 { - rv := make(search.DocumentMatchCollection, tksc.results.Len()-tksc.skip) - i := 0 - skipped := 0 - for e := tksc.results.Back(); e != nil; e = e.Prev() { - if skipped < tksc.skip { - skipped++ - continue - } - rv[i] = e.Value.(*search.DocumentMatch) - i++ - } - return rv - } - return search.DocumentMatchCollection{} -} - -func (tksc *TopScoreCollector) SetFacetsBuilder(facetsBuilder *search.FacetsBuilder) { - tksc.facetsBuilder = facetsBuilder -} - -func (tksc *TopScoreCollector) FacetResults() search.FacetResults { - if tksc.facetsBuilder != nil { - return tksc.facetsBuilder.Results() - } - return search.FacetResults{} -} diff --git a/vendor/github.com/blevesearch/bleve/search/collectors/heap.go b/vendor/github.com/blevesearch/bleve/search/collectors/heap.go new file mode 100644 index 0000000..3f62867 --- /dev/null +++ b/vendor/github.com/blevesearch/bleve/search/collectors/heap.go @@ -0,0 +1,86 @@ +// Copyright (c) 2014 Couchbase, Inc. +// Licensed under the Apache License, Version 2.0 (the "License"); you may not use this file +// except in compliance with the License. You may obtain a copy of the License at +// http://www.apache.org/licenses/LICENSE-2.0 +// Unless required by applicable law or agreed to in writing, software distributed under the +// License is distributed on an "AS IS" BASIS, WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, +// either express or implied. See the License for the specific language governing permissions +// and limitations under the License. + +package collectors + +import ( + "container/heap" + + "github.com/blevesearch/bleve/search" +) + +type collectStoreHeap struct { + heap search.DocumentMatchCollection + compare collectorCompare +} + +func newStoreHeap(cap int, compare collectorCompare) *collectStoreHeap { + rv := &collectStoreHeap{ + heap: make(search.DocumentMatchCollection, 0, cap), + compare: compare, + } + heap.Init(rv) + return rv +} + +func (c *collectStoreHeap) Add(doc *search.DocumentMatch) { + heap.Push(c, doc) +} + +func (c *collectStoreHeap) RemoveLast() *search.DocumentMatch { + return heap.Pop(c).(*search.DocumentMatch) +} + +func (c *collectStoreHeap) Final(skip int, fixup collectorFixup) (search.DocumentMatchCollection, error) { + count := c.Len() + size := count - skip + if size <= 0 { + return make(search.DocumentMatchCollection, 0), nil + } + rv := make(search.DocumentMatchCollection, size) + for count > 0 { + count-- + + if count >= skip { + size-- + doc := heap.Pop(c).(*search.DocumentMatch) + rv[size] = doc + err := fixup(doc) + if err != nil { + return nil, err + } + } + } + return rv, nil +} + +// heap interface implementation + +func (c *collectStoreHeap) Len() int { + return len(c.heap) +} + +func (c *collectStoreHeap) Less(i, j int) bool { + so := c.compare(c.heap[i], c.heap[j]) + return -so < 0 +} + +func (c *collectStoreHeap) Swap(i, j int) { + c.heap[i], c.heap[j] = c.heap[j], c.heap[i] +} + +func (c *collectStoreHeap) Push(x interface{}) { + c.heap = append(c.heap, x.(*search.DocumentMatch)) +} + +func (c *collectStoreHeap) Pop() interface{} { + var rv *search.DocumentMatch + rv, c.heap = c.heap[len(c.heap)-1], c.heap[:len(c.heap)-1] + return rv +} diff --git a/vendor/github.com/blevesearch/bleve/search/collectors/list.go b/vendor/github.com/blevesearch/bleve/search/collectors/list.go new file mode 100644 index 0000000..d3f4941 --- /dev/null +++ b/vendor/github.com/blevesearch/bleve/search/collectors/list.go @@ -0,0 +1,73 @@ +// Copyright (c) 2014 Couchbase, Inc. +// Licensed under the Apache License, Version 2.0 (the "License"); you may not use this file +// except in compliance with the License. You may obtain a copy of the License at +// http://www.apache.org/licenses/LICENSE-2.0 +// Unless required by applicable law or agreed to in writing, software distributed under the +// License is distributed on an "AS IS" BASIS, WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, +// either express or implied. See the License for the specific language governing permissions +// and limitations under the License. + +package collectors + +import ( + "container/list" + + "github.com/blevesearch/bleve/search" +) + +type collectStoreList struct { + results *list.List + compare collectorCompare +} + +func newStoreList(cap int, compare collectorCompare) *collectStoreList { + rv := &collectStoreList{ + results: list.New(), + compare: compare, + } + + return rv +} + +func (c *collectStoreList) Add(doc *search.DocumentMatch) { + for e := c.results.Front(); e != nil; e = e.Next() { + curr := e.Value.(*search.DocumentMatch) + if c.compare(doc, curr) >= 0 { + c.results.InsertBefore(doc, e) + return + } + } + // if we got to the end, we still have to add it + c.results.PushBack(doc) +} + +func (c *collectStoreList) RemoveLast() *search.DocumentMatch { + return c.results.Remove(c.results.Front()).(*search.DocumentMatch) +} + +func (c *collectStoreList) Final(skip int, fixup collectorFixup) (search.DocumentMatchCollection, error) { + if c.results.Len()-skip > 0 { + rv := make(search.DocumentMatchCollection, c.results.Len()-skip) + i := 0 + skipped := 0 + for e := c.results.Back(); e != nil; e = e.Prev() { + if skipped < skip { + skipped++ + continue + } + + rv[i] = e.Value.(*search.DocumentMatch) + err := fixup(rv[i]) + if err != nil { + return nil, err + } + i++ + } + return rv, nil + } + return search.DocumentMatchCollection{}, nil +} + +func (c *collectStoreList) Len() int { + return c.results.Len() +} diff --git a/vendor/github.com/blevesearch/bleve/search/collectors/slice.go b/vendor/github.com/blevesearch/bleve/search/collectors/slice.go new file mode 100644 index 0000000..26b29c7 --- /dev/null +++ b/vendor/github.com/blevesearch/bleve/search/collectors/slice.go @@ -0,0 +1,63 @@ +// Copyright (c) 2014 Couchbase, Inc. +// Licensed under the Apache License, Version 2.0 (the "License"); you may not use this file +// except in compliance with the License. You may obtain a copy of the License at +// http://www.apache.org/licenses/LICENSE-2.0 +// Unless required by applicable law or agreed to in writing, software distributed under the +// License is distributed on an "AS IS" BASIS, WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, +// either express or implied. See the License for the specific language governing permissions +// and limitations under the License. + +package collectors + +import "github.com/blevesearch/bleve/search" + +type collectStoreSlice struct { + slice search.DocumentMatchCollection + compare collectorCompare +} + +func newStoreSlice(cap int, compare collectorCompare) *collectStoreSlice { + rv := &collectStoreSlice{ + slice: make(search.DocumentMatchCollection, 0, cap), + compare: compare, + } + return rv +} + +func (c *collectStoreSlice) Add(doc *search.DocumentMatch) { + // find where to insert, starting at end (lowest) + i := len(c.slice) + for ; i > 0; i-- { + cmp := c.compare(doc, c.slice[i-1]) + if cmp >= 0 { + break + } + } + // insert at i + c.slice = append(c.slice, nil) + copy(c.slice[i+1:], c.slice[i:]) + c.slice[i] = doc +} + +func (c *collectStoreSlice) RemoveLast() *search.DocumentMatch { + var rv *search.DocumentMatch + rv, c.slice = c.slice[len(c.slice)-1], c.slice[:len(c.slice)-1] + return rv +} + +func (c *collectStoreSlice) Final(skip int, fixup collectorFixup) (search.DocumentMatchCollection, error) { + for i := skip; i < len(c.slice); i++ { + err := fixup(c.slice[i]) + if err != nil { + return nil, err + } + } + if skip <= len(c.slice) { + return c.slice[skip:], nil + } + return search.DocumentMatchCollection{}, nil +} + +func (c *collectStoreSlice) Len() int { + return len(c.slice) +} diff --git a/vendor/github.com/blevesearch/bleve/search/collectors/topn.go b/vendor/github.com/blevesearch/bleve/search/collectors/topn.go new file mode 100644 index 0000000..4c5a566 --- /dev/null +++ b/vendor/github.com/blevesearch/bleve/search/collectors/topn.go @@ -0,0 +1,264 @@ +// Copyright (c) 2014 Couchbase, Inc. +// Licensed under the Apache License, Version 2.0 (the "License"); you may not use this file +// except in compliance with the License. You may obtain a copy of the License at +// http://www.apache.org/licenses/LICENSE-2.0 +// Unless required by applicable law or agreed to in writing, software distributed under the +// License is distributed on an "AS IS" BASIS, WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, +// either express or implied. See the License for the specific language governing permissions +// and limitations under the License. + +package collectors + +import ( + "time" + + "github.com/blevesearch/bleve/index" + "github.com/blevesearch/bleve/search" + "golang.org/x/net/context" +) + +// PreAllocSizeSkipCap will cap preallocation to this amount when +// size+skip exceeds this value +var PreAllocSizeSkipCap = 1000 + +type collectorCompare func(i, j *search.DocumentMatch) int + +type collectorFixup func(d *search.DocumentMatch) error + +// TopNCollector collects the top N hits, optionally skipping some results +type TopNCollector struct { + size int + skip int + total uint64 + maxScore float64 + took time.Duration + sort search.SortOrder + results search.DocumentMatchCollection + facetsBuilder *search.FacetsBuilder + + store *collectStoreSlice + + needDocIds bool + neededFields []string + cachedScoring []bool + cachedDesc []bool + + lowestMatchOutsideResults *search.DocumentMatch +} + +// CheckDoneEvery controls how frequently we check the context deadline +const CheckDoneEvery = uint64(1024) + +// NewTopNCollector builds a collector to find the top 'size' hits +// skipping over the first 'skip' hits +// ordering hits by the provided sort order +func NewTopNCollector(size int, skip int, sort search.SortOrder) *TopNCollector { + hc := &TopNCollector{size: size, skip: skip, sort: sort} + + // pre-allocate space on the store to avoid reslicing + // unless the size + skip is too large, then cap it + // everything should still work, just reslices as necessary + backingSize := size + skip + 1 + if size+skip > PreAllocSizeSkipCap { + backingSize = PreAllocSizeSkipCap + 1 + } + + hc.store = newStoreSlice(backingSize, func(i, j *search.DocumentMatch) int { + return hc.sort.Compare(hc.cachedScoring, hc.cachedDesc, i, j) + }) + + // these lookups traverse an interface, so do once up-front + if sort.RequiresDocID() { + hc.needDocIds = true + } + hc.neededFields = sort.RequiredFields() + hc.cachedScoring = sort.CacheIsScore() + hc.cachedDesc = sort.CacheDescending() + + return hc +} + +// Collect goes to the index to find the matching documents +func (hc *TopNCollector) Collect(ctx context.Context, searcher search.Searcher, reader index.IndexReader) error { + startTime := time.Now() + var err error + var next *search.DocumentMatch + + // pre-allocate enough space in the DocumentMatchPool + // unless the size + skip is too large, then cap it + // everything should still work, just allocates DocumentMatches on demand + backingSize := hc.size + hc.skip + 1 + if hc.size+hc.skip > PreAllocSizeSkipCap { + backingSize = PreAllocSizeSkipCap + 1 + } + searchContext := &search.SearchContext{ + DocumentMatchPool: search.NewDocumentMatchPool(backingSize+searcher.DocumentMatchPoolSize(), len(hc.sort)), + } + + select { + case <-ctx.Done(): + return ctx.Err() + default: + next, err = searcher.Next(searchContext) + } + for err == nil && next != nil { + if hc.total%CheckDoneEvery == 0 { + select { + case <-ctx.Done(): + return ctx.Err() + default: + } + } + if hc.facetsBuilder != nil { + err = hc.facetsBuilder.Update(next) + if err != nil { + break + } + } + + err = hc.collectSingle(searchContext, reader, next) + if err != nil { + break + } + + next, err = searcher.Next(searchContext) + } + // compute search duration + hc.took = time.Since(startTime) + if err != nil { + return err + } + // finalize actual results + err = hc.finalizeResults(reader) + if err != nil { + return err + } + return nil +} + +var sortByScoreOpt = []string{"_score"} + +func (hc *TopNCollector) collectSingle(ctx *search.SearchContext, reader index.IndexReader, d *search.DocumentMatch) error { + // increment total hits + hc.total++ + d.HitNumber = hc.total + + // update max score + if d.Score > hc.maxScore { + hc.maxScore = d.Score + } + + var err error + // see if we need to load ID (at this early stage, for example to sort on it) + if hc.needDocIds { + d.ID, err = reader.ExternalID(d.IndexInternalID) + if err != nil { + return err + } + } + + // see if we need to load the stored fields + if len(hc.neededFields) > 0 { + // find out which fields haven't been loaded yet + fieldsToLoad := d.CachedFieldTerms.FieldsNotYetCached(hc.neededFields) + // look them up + fieldTerms, err := reader.DocumentFieldTerms(d.IndexInternalID, fieldsToLoad) + if err != nil { + return err + } + // cache these as well + if d.CachedFieldTerms == nil { + d.CachedFieldTerms = make(map[string][]string) + } + d.CachedFieldTerms.Merge(fieldTerms) + } + + // compute this hits sort value + if len(hc.sort) == 1 && hc.cachedScoring[0] { + d.Sort = sortByScoreOpt + } else { + hc.sort.Value(d) + } + + // optimization, we track lowest sorting hit already removed from heap + // with this one comparision, we can avoid all heap operations if + // this hit would have been added and then immediately removed + if hc.lowestMatchOutsideResults != nil { + cmp := hc.sort.Compare(hc.cachedScoring, hc.cachedDesc, d, hc.lowestMatchOutsideResults) + if cmp >= 0 { + // this hit can't possibly be in the result set, so avoid heap ops + ctx.DocumentMatchPool.Put(d) + return nil + } + } + + hc.store.Add(d) + if hc.store.Len() > hc.size+hc.skip { + removed := hc.store.RemoveLast() + if hc.lowestMatchOutsideResults == nil { + hc.lowestMatchOutsideResults = removed + } else { + cmp := hc.sort.Compare(hc.cachedScoring, hc.cachedDesc, removed, hc.lowestMatchOutsideResults) + if cmp < 0 { + tmp := hc.lowestMatchOutsideResults + hc.lowestMatchOutsideResults = removed + ctx.DocumentMatchPool.Put(tmp) + } + } + } + + return nil +} + +// SetFacetsBuilder registers a facet builder for this collector +func (hc *TopNCollector) SetFacetsBuilder(facetsBuilder *search.FacetsBuilder) { + hc.facetsBuilder = facetsBuilder +} + +// finalizeResults starts with the heap containing the final top size+skip +// it now throws away the results to be skipped +// and does final doc id lookup (if necessary) +func (hc *TopNCollector) finalizeResults(r index.IndexReader) error { + var err error + hc.results, err = hc.store.Final(hc.skip, func(doc *search.DocumentMatch) error { + if doc.ID == "" { + // look up the id since we need it for lookup + var err error + doc.ID, err = r.ExternalID(doc.IndexInternalID) + if err != nil { + return err + } + } + return nil + }) + + return err +} + +// Results returns the collected hits +func (hc *TopNCollector) Results() search.DocumentMatchCollection { + return hc.results +} + +// Total returns the total number of hits +func (hc *TopNCollector) Total() uint64 { + return hc.total +} + +// MaxScore returns the maximum score seen across all the hits +func (hc *TopNCollector) MaxScore() float64 { + return hc.maxScore +} + +// Took returns the time spent collecting hits +func (hc *TopNCollector) Took() time.Duration { + return hc.took +} + +// FacetResults returns the computed facets results +func (hc *TopNCollector) FacetResults() search.FacetResults { + if hc.facetsBuilder != nil { + return hc.facetsBuilder.Results() + } + return search.FacetResults{} +} diff --git a/vendor/github.com/blevesearch/bleve/search/facets/facet_builder_datetime.go b/vendor/github.com/blevesearch/bleve/search/facets/facet_builder_datetime.go index 3898146..7e94d52 100644 --- a/vendor/github.com/blevesearch/bleve/search/facets/facet_builder_datetime.go +++ b/vendor/github.com/blevesearch/bleve/search/facets/facet_builder_datetime.go @@ -49,6 +49,10 @@ func (fb *DateTimeFacetBuilder) AddRange(name string, start, end time.Time) { fb.ranges[name] = &r } +func (fb *DateTimeFacetBuilder) Field() string { + return fb.field +} + func (fb *DateTimeFacetBuilder) Update(ft index.FieldTerms) { terms, ok := ft[fb.field] if ok { diff --git a/vendor/github.com/blevesearch/bleve/search/facets/facet_builder_numeric.go b/vendor/github.com/blevesearch/bleve/search/facets/facet_builder_numeric.go index f5acfb0..a1ac311 100644 --- a/vendor/github.com/blevesearch/bleve/search/facets/facet_builder_numeric.go +++ b/vendor/github.com/blevesearch/bleve/search/facets/facet_builder_numeric.go @@ -48,6 +48,10 @@ func (fb *NumericFacetBuilder) AddRange(name string, min, max *float64) { fb.ranges[name] = &r } +func (fb *NumericFacetBuilder) Field() string { + return fb.field +} + func (fb *NumericFacetBuilder) Update(ft index.FieldTerms) { terms, ok := ft[fb.field] if ok { diff --git a/vendor/github.com/blevesearch/bleve/search/facets/facet_builder_terms.go b/vendor/github.com/blevesearch/bleve/search/facets/facet_builder_terms.go index 35c56f2..4488139 100644 --- a/vendor/github.com/blevesearch/bleve/search/facets/facet_builder_terms.go +++ b/vendor/github.com/blevesearch/bleve/search/facets/facet_builder_terms.go @@ -32,6 +32,10 @@ func NewTermsFacetBuilder(field string, size int) *TermsFacetBuilder { } } +func (fb *TermsFacetBuilder) Field() string { + return fb.field +} + func (fb *TermsFacetBuilder) Update(ft index.FieldTerms) { terms, ok := ft[fb.field] if ok { diff --git a/vendor/github.com/blevesearch/bleve/search/facets_builder.go b/vendor/github.com/blevesearch/bleve/search/facets_builder.go index f41be29..66f96a6 100644 --- a/vendor/github.com/blevesearch/bleve/search/facets_builder.go +++ b/vendor/github.com/blevesearch/bleve/search/facets_builder.go @@ -18,6 +18,7 @@ import ( type FacetBuilder interface { Update(index.FieldTerms) Result() *FacetResult + Field() string } type FacetsBuilder struct { @@ -37,12 +38,27 @@ func (fb *FacetsBuilder) Add(name string, facetBuilder FacetBuilder) { } func (fb *FacetsBuilder) Update(docMatch *DocumentMatch) error { - fieldTerms, err := fb.indexReader.DocumentFieldTerms(docMatch.ID) - if err != nil { - return err + var fields []string + for _, facetBuilder := range fb.facets { + fields = append(fields, facetBuilder.Field()) + } + + if len(fields) > 0 { + // find out which fields haven't been loaded yet + fieldsToLoad := docMatch.CachedFieldTerms.FieldsNotYetCached(fields) + // look them up + fieldTerms, err := fb.indexReader.DocumentFieldTerms(docMatch.IndexInternalID, fieldsToLoad) + if err != nil { + return err + } + // cache these as well + if docMatch.CachedFieldTerms == nil { + docMatch.CachedFieldTerms = make(map[string][]string) + } + docMatch.CachedFieldTerms.Merge(fieldTerms) } for _, facetBuilder := range fb.facets { - facetBuilder.Update(fieldTerms) + facetBuilder.Update(docMatch.CachedFieldTerms) } return nil } diff --git a/vendor/github.com/blevesearch/bleve/search/pool.go b/vendor/github.com/blevesearch/bleve/search/pool.go new file mode 100644 index 0000000..5600f48 --- /dev/null +++ b/vendor/github.com/blevesearch/bleve/search/pool.go @@ -0,0 +1,71 @@ +// Copyright (c) 2014 Couchbase, Inc. +// Licensed under the Apache License, Version 2.0 (the "License"); you may not use this file +// except in compliance with the License. You may obtain a copy of the License at +// http://www.apache.org/licenses/LICENSE-2.0 +// Unless required by applicable law or agreed to in writing, software distributed under the +// License is distributed on an "AS IS" BASIS, WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, +// either express or implied. See the License for the specific language governing permissions +// and limitations under the License. + +package search + +// DocumentMatchPoolTooSmall is a callback function that can be executed +// when the DocumentMatchPool does not have sufficient capacity +// By default we just perform just-in-time allocation, but you could log +// a message, or panic, etc. +type DocumentMatchPoolTooSmall func(p *DocumentMatchPool) *DocumentMatch + +// DocumentMatchPool manages use/re-use of DocumentMatch instances +// it pre-allocates space from a single large block with the expected +// number of instances. It is not thread-safe as currently all +// aspects of search take place in a single goroutine. +type DocumentMatchPool struct { + avail DocumentMatchCollection + TooSmall DocumentMatchPoolTooSmall +} + +func defaultDocumentMatchPoolTooSmall(p *DocumentMatchPool) *DocumentMatch { + return &DocumentMatch{} +} + +// NewDocumentMatchPool will build a DocumentMatchPool with memory +// pre-allocated to accomodate the requested number of DocumentMatch +// instances +func NewDocumentMatchPool(size, sortsize int) *DocumentMatchPool { + avail := make(DocumentMatchCollection, 0, size) + // pre-allocate the expected number of instances + startBlock := make([]DocumentMatch, size) + // make these initial instances available + for i := range startBlock { + startBlock[i].Sort = make([]string, 0, sortsize) + avail = append(avail, &startBlock[i]) + } + return &DocumentMatchPool{ + avail: avail, + TooSmall: defaultDocumentMatchPoolTooSmall, + } +} + +// Get returns an available DocumentMatch from the pool +// if the pool was not allocated with sufficient size, an allocation will +// occur to satisfy this request. As a side-effect this will grow the size +// of the pool. +func (p *DocumentMatchPool) Get() *DocumentMatch { + var rv *DocumentMatch + if len(p.avail) > 0 { + rv, p.avail = p.avail[len(p.avail)-1], p.avail[:len(p.avail)-1] + } else { + rv = p.TooSmall(p) + } + return rv +} + +// Put returns a DocumentMatch to the pool +func (p *DocumentMatchPool) Put(d *DocumentMatch) { + if d == nil { + return + } + // reset DocumentMatch before returning it to available pool + d.Reset() + p.avail = append(p.avail, d) +} diff --git a/vendor/github.com/blevesearch/bleve/search/scorers/scorer_conjunction.go b/vendor/github.com/blevesearch/bleve/search/scorers/scorer_conjunction.go index 422f282..c941c69 100644 --- a/vendor/github.com/blevesearch/bleve/search/scorers/scorer_conjunction.go +++ b/vendor/github.com/blevesearch/bleve/search/scorers/scorer_conjunction.go @@ -23,11 +23,7 @@ func NewConjunctionQueryScorer(explain bool) *ConjunctionQueryScorer { } } -func (s *ConjunctionQueryScorer) Score(constituents []*search.DocumentMatch) *search.DocumentMatch { - rv := search.DocumentMatch{ - ID: constituents[0].ID, - } - +func (s *ConjunctionQueryScorer) Score(ctx *search.SearchContext, constituents []*search.DocumentMatch) *search.DocumentMatch { var sum float64 var childrenExplanations []*search.Explanation if s.explain { @@ -44,16 +40,21 @@ func (s *ConjunctionQueryScorer) Score(constituents []*search.DocumentMatch) *se locations = append(locations, docMatch.Locations) } } - rv.Score = sum + newScore := sum + var newExpl *search.Explanation if s.explain { - rv.Expl = &search.Explanation{Value: sum, Message: "sum of:", Children: childrenExplanations} + newExpl = &search.Explanation{Value: sum, Message: "sum of:", Children: childrenExplanations} } + // reuse constituents[0] as the return value + rv := constituents[0] + rv.Score = newScore + rv.Expl = newExpl if len(locations) == 1 { rv.Locations = locations[0] } else if len(locations) > 1 { rv.Locations = search.MergeLocations(locations) } - return &rv + return rv } diff --git a/vendor/github.com/blevesearch/bleve/search/scorers/scorer_constant.go b/vendor/github.com/blevesearch/bleve/search/scorers/scorer_constant.go index 1434bd5..adbaf47 100644 --- a/vendor/github.com/blevesearch/bleve/search/scorers/scorer_constant.go +++ b/vendor/github.com/blevesearch/bleve/search/scorers/scorer_constant.go @@ -12,6 +12,7 @@ package scorers import ( "fmt" + "github.com/blevesearch/bleve/index" "github.com/blevesearch/bleve/search" ) @@ -64,7 +65,7 @@ func (s *ConstantScorer) SetQueryNorm(qnorm float64) { } } -func (s *ConstantScorer) Score(id string) *search.DocumentMatch { +func (s *ConstantScorer) Score(ctx *search.SearchContext, id index.IndexInternalID) *search.DocumentMatch { var scoreExplanation *search.Explanation score := s.constant @@ -91,13 +92,12 @@ func (s *ConstantScorer) Score(id string) *search.DocumentMatch { } } - rv := search.DocumentMatch{ - ID: id, - Score: score, - } + rv := ctx.DocumentMatchPool.Get() + rv.IndexInternalID = id + rv.Score = score if s.explain { rv.Expl = scoreExplanation } - return &rv + return rv } diff --git a/vendor/github.com/blevesearch/bleve/search/scorers/scorer_disjunction.go b/vendor/github.com/blevesearch/bleve/search/scorers/scorer_disjunction.go index 00bc8cd..2aca054 100644 --- a/vendor/github.com/blevesearch/bleve/search/scorers/scorer_disjunction.go +++ b/vendor/github.com/blevesearch/bleve/search/scorers/scorer_disjunction.go @@ -25,18 +25,14 @@ func NewDisjunctionQueryScorer(explain bool) *DisjunctionQueryScorer { } } -func (s *DisjunctionQueryScorer) Score(constituents []*search.DocumentMatch, countMatch, countTotal int) *search.DocumentMatch { - rv := search.DocumentMatch{ - ID: constituents[0].ID, - } - +func (s *DisjunctionQueryScorer) Score(ctx *search.SearchContext, constituents []*search.DocumentMatch, countMatch, countTotal int) *search.DocumentMatch { var sum float64 var childrenExplanations []*search.Explanation if s.explain { childrenExplanations = make([]*search.Explanation, len(constituents)) } - locations := []search.FieldTermLocationMap{} + var locations []search.FieldTermLocationMap for i, docMatch := range constituents { sum += docMatch.Score if s.explain { @@ -53,19 +49,24 @@ func (s *DisjunctionQueryScorer) Score(constituents []*search.DocumentMatch, cou } coord := float64(countMatch) / float64(countTotal) - rv.Score = sum * coord + newScore := sum * coord + var newExpl *search.Explanation if s.explain { ce := make([]*search.Explanation, 2) ce[0] = rawExpl ce[1] = &search.Explanation{Value: coord, Message: fmt.Sprintf("coord(%d/%d)", countMatch, countTotal)} - rv.Expl = &search.Explanation{Value: rv.Score, Message: "product of:", Children: ce} + newExpl = &search.Explanation{Value: newScore, Message: "product of:", Children: ce} } + // reuse constituents[0] as the return value + rv := constituents[0] + rv.Score = newScore + rv.Expl = newExpl if len(locations) == 1 { rv.Locations = locations[0] } else if len(locations) > 1 { rv.Locations = search.MergeLocations(locations) } - return &rv + return rv } diff --git a/vendor/github.com/blevesearch/bleve/search/scorers/scorer_term.go b/vendor/github.com/blevesearch/bleve/search/scorers/scorer_term.go index 0a0518d..4304d06 100644 --- a/vendor/github.com/blevesearch/bleve/search/scorers/scorer_term.go +++ b/vendor/github.com/blevesearch/bleve/search/scorers/scorer_term.go @@ -83,7 +83,7 @@ func (s *TermQueryScorer) SetQueryNorm(qnorm float64) { } } -func (s *TermQueryScorer) Score(termMatch *index.TermFieldDoc) *search.DocumentMatch { +func (s *TermQueryScorer) Score(ctx *search.SearchContext, termMatch *index.TermFieldDoc) *search.DocumentMatch { var scoreExplanation *search.Explanation // need to compute score @@ -128,10 +128,9 @@ func (s *TermQueryScorer) Score(termMatch *index.TermFieldDoc) *search.DocumentM } } - rv := search.DocumentMatch{ - ID: termMatch.ID, - Score: score, - } + rv := ctx.DocumentMatchPool.Get() + rv.IndexInternalID = append(rv.IndexInternalID, termMatch.ID...) + rv.Score = score if s.explain { rv.Expl = scoreExplanation } @@ -172,5 +171,5 @@ func (s *TermQueryScorer) Score(termMatch *index.TermFieldDoc) *search.DocumentM } - return &rv + return rv } diff --git a/vendor/github.com/blevesearch/bleve/search/scorers/sqrt_cache.go b/vendor/github.com/blevesearch/bleve/search/scorers/sqrt_cache.go index f93d27c..d444c25 100644 --- a/vendor/github.com/blevesearch/bleve/search/scorers/sqrt_cache.go +++ b/vendor/github.com/blevesearch/bleve/search/scorers/sqrt_cache.go @@ -13,12 +13,12 @@ import ( "math" ) -var SqrtCache map[int]float64 +var SqrtCache []float64 const MaxSqrtCache = 64 func init() { - SqrtCache = make(map[int]float64, MaxSqrtCache) + SqrtCache = make([]float64, MaxSqrtCache) for i := 0; i < MaxSqrtCache; i++ { SqrtCache[i] = math.Sqrt(float64(i)) } diff --git a/vendor/github.com/blevesearch/bleve/search/search.go b/vendor/github.com/blevesearch/bleve/search/search.go index cc4b175..5e43b74 100644 --- a/vendor/github.com/blevesearch/bleve/search/search.go +++ b/vendor/github.com/blevesearch/bleve/search/search.go @@ -9,6 +9,13 @@ package search +import ( + "fmt" + + "github.com/blevesearch/bleve/document" + "github.com/blevesearch/bleve/index" +) + type Location struct { Pos float64 `json:"pos"` Start float64 `json:"start"` @@ -51,17 +58,29 @@ type FieldTermLocationMap map[string]TermLocationMap type FieldFragmentMap map[string][]string type DocumentMatch struct { - Index string `json:"index,omitempty"` - ID string `json:"id"` - Score float64 `json:"score"` - Expl *Explanation `json:"explanation,omitempty"` - Locations FieldTermLocationMap `json:"locations,omitempty"` - Fragments FieldFragmentMap `json:"fragments,omitempty"` + Index string `json:"index,omitempty"` + ID string `json:"id"` + IndexInternalID index.IndexInternalID `json:"-"` + Score float64 `json:"score"` + Expl *Explanation `json:"explanation,omitempty"` + Locations FieldTermLocationMap `json:"locations,omitempty"` + Fragments FieldFragmentMap `json:"fragments,omitempty"` + Sort []string `json:"sort,omitempty"` // Fields contains the values for document fields listed in // SearchRequest.Fields. Text fields are returned as strings, numeric // fields as float64s and date fields as time.RFC3339 formatted strings. Fields map[string]interface{} `json:"fields,omitempty"` + + // as we learn field terms, we can cache important ones for later use + // for example, sorting and building facets need these values + CachedFieldTerms index.FieldTerms `json:"-"` + + // if we load the document for this hit, remember it so we dont load again + Document *document.Document `json:"-"` + + // used to maintain natural index order + HitNumber uint64 `json:"-"` } func (dm *DocumentMatch) AddFieldValue(name string, value interface{}) { @@ -85,6 +104,25 @@ func (dm *DocumentMatch) AddFieldValue(name string, value interface{}) { dm.Fields[name] = valSlice } +// Reset allows an already allocated DocumentMatch to be reused +func (dm *DocumentMatch) Reset() *DocumentMatch { + // remember the []byte used for the IndexInternalID + indexInternalID := dm.IndexInternalID + // remember the []interface{} used for sort + sort := dm.Sort + // idiom to copy over from empty DocumentMatch (0 allocations) + *dm = DocumentMatch{} + // reuse the []byte already allocated (and reset len to 0) + dm.IndexInternalID = indexInternalID[:0] + // reuse the []interface{} already allocated (and reset len to 0) + dm.Sort = sort[:0] + return dm +} + +func (dm *DocumentMatch) String() string { + return fmt.Sprintf("[%s-%f]", string(dm.IndexInternalID), dm.Score) +} + type DocumentMatchCollection []*DocumentMatch func (c DocumentMatchCollection) Len() int { return len(c) } @@ -92,11 +130,18 @@ func (c DocumentMatchCollection) Swap(i, j int) { c[i], c[j] = c[j], c[i] } func (c DocumentMatchCollection) Less(i, j int) bool { return c[i].Score > c[j].Score } type Searcher interface { - Next() (*DocumentMatch, error) - Advance(ID string) (*DocumentMatch, error) + Next(ctx *SearchContext) (*DocumentMatch, error) + Advance(ctx *SearchContext, ID index.IndexInternalID) (*DocumentMatch, error) Close() error Weight() float64 SetQueryNorm(float64) Count() uint64 Min() int + + DocumentMatchPoolSize() int +} + +// SearchContext represents the context around a single search +type SearchContext struct { + DocumentMatchPool *DocumentMatchPool } diff --git a/vendor/github.com/blevesearch/bleve/search/searchers/search_boolean.go b/vendor/github.com/blevesearch/bleve/search/searchers/search_boolean.go index bf44e02..8a94929 100644 --- a/vendor/github.com/blevesearch/bleve/search/searchers/search_boolean.go +++ b/vendor/github.com/blevesearch/bleve/search/searchers/search_boolean.go @@ -18,7 +18,6 @@ import ( ) type BooleanSearcher struct { - initialized bool indexReader index.IndexReader mustSearcher search.Searcher shouldSearcher search.Searcher @@ -27,9 +26,11 @@ type BooleanSearcher struct { currMust *search.DocumentMatch currShould *search.DocumentMatch currMustNot *search.DocumentMatch - currentID string + currentID index.IndexInternalID min uint64 scorer *scorers.ConjunctionQueryScorer + matches []*search.DocumentMatch + initialized bool } func NewBooleanSearcher(indexReader index.IndexReader, mustSearcher search.Searcher, shouldSearcher search.Searcher, mustNotSearcher search.Searcher, explain bool) (*BooleanSearcher, error) { @@ -40,6 +41,7 @@ func NewBooleanSearcher(indexReader index.IndexReader, mustSearcher search.Searc shouldSearcher: shouldSearcher, mustNotSearcher: mustNotSearcher, scorer: scorers.NewConjunctionQueryScorer(explain), + matches: make([]*search.DocumentMatch, 2), } rv.computeQueryNorm() return &rv, nil @@ -66,63 +68,78 @@ func (s *BooleanSearcher) computeQueryNorm() { } } -func (s *BooleanSearcher) initSearchers() error { +func (s *BooleanSearcher) initSearchers(ctx *search.SearchContext) error { var err error // get all searchers pointing at their first match if s.mustSearcher != nil { - s.currMust, err = s.mustSearcher.Next() + if s.currMust != nil { + ctx.DocumentMatchPool.Put(s.currMust) + } + s.currMust, err = s.mustSearcher.Next(ctx) if err != nil { return err } } if s.shouldSearcher != nil { - s.currShould, err = s.shouldSearcher.Next() + if s.currShould != nil { + ctx.DocumentMatchPool.Put(s.currShould) + } + s.currShould, err = s.shouldSearcher.Next(ctx) if err != nil { return err } } if s.mustNotSearcher != nil { - s.currMustNot, err = s.mustNotSearcher.Next() + if s.currMustNot != nil { + ctx.DocumentMatchPool.Put(s.currMustNot) + } + s.currMustNot, err = s.mustNotSearcher.Next(ctx) if err != nil { return err } } if s.mustSearcher != nil && s.currMust != nil { - s.currentID = s.currMust.ID + s.currentID = s.currMust.IndexInternalID } else if s.mustSearcher == nil && s.currShould != nil { - s.currentID = s.currShould.ID + s.currentID = s.currShould.IndexInternalID } else { - s.currentID = "" + s.currentID = nil } s.initialized = true return nil } -func (s *BooleanSearcher) advanceNextMust() error { +func (s *BooleanSearcher) advanceNextMust(ctx *search.SearchContext, skipReturn *search.DocumentMatch) error { var err error if s.mustSearcher != nil { - s.currMust, err = s.mustSearcher.Next() + if s.currMust != skipReturn { + ctx.DocumentMatchPool.Put(s.currMust) + } + s.currMust, err = s.mustSearcher.Next(ctx) if err != nil { return err } - } else if s.mustSearcher == nil { - s.currShould, err = s.shouldSearcher.Next() + } else { + if s.currShould != skipReturn { + ctx.DocumentMatchPool.Put(s.currShould) + } + s.currShould, err = s.shouldSearcher.Next(ctx) if err != nil { return err } } if s.mustSearcher != nil && s.currMust != nil { - s.currentID = s.currMust.ID + s.currentID = s.currMust.IndexInternalID } else if s.mustSearcher == nil && s.currShould != nil { - s.currentID = s.currShould.ID + s.currentID = s.currShould.IndexInternalID } else { - s.currentID = "" + s.currentID = nil } return nil } @@ -148,10 +165,10 @@ func (s *BooleanSearcher) SetQueryNorm(qnorm float64) { } } -func (s *BooleanSearcher) Next() (*search.DocumentMatch, error) { +func (s *BooleanSearcher) Next(ctx *search.SearchContext) (*search.DocumentMatch, error) { if !s.initialized { - err := s.initSearchers() + err := s.initSearchers(ctx) if err != nil { return nil, err } @@ -160,94 +177,104 @@ func (s *BooleanSearcher) Next() (*search.DocumentMatch, error) { var err error var rv *search.DocumentMatch - for s.currentID != "" { - if s.currMustNot != nil && s.currMustNot.ID < s.currentID { - // advance must not searcher to our candidate entry - s.currMustNot, err = s.mustNotSearcher.Advance(s.currentID) - if err != nil { - return nil, err - } - if s.currMustNot != nil && s.currMustNot.ID == s.currentID { + for s.currentID != nil { + if s.currMustNot != nil { + cmp := s.currMustNot.IndexInternalID.Compare(s.currentID) + if cmp < 0 { + ctx.DocumentMatchPool.Put(s.currMustNot) + // advance must not searcher to our candidate entry + s.currMustNot, err = s.mustNotSearcher.Advance(ctx, s.currentID) + if err != nil { + return nil, err + } + if s.currMustNot != nil && s.currMustNot.IndexInternalID.Equals(s.currentID) { + // the candidate is excluded + err = s.advanceNextMust(ctx, nil) + if err != nil { + return nil, err + } + continue + } + } else if cmp == 0 { // the candidate is excluded - err = s.advanceNextMust() + err = s.advanceNextMust(ctx, nil) if err != nil { return nil, err } continue } - } else if s.currMustNot != nil && s.currMustNot.ID == s.currentID { - // the candidate is excluded - err = s.advanceNextMust() - if err != nil { - return nil, err - } - continue } - if s.currShould != nil && s.currShould.ID < s.currentID { + shouldCmpOrNil := 1 // NOTE: shouldCmp will also be 1 when currShould == nil. + if s.currShould != nil { + shouldCmpOrNil = s.currShould.IndexInternalID.Compare(s.currentID) + } + + if shouldCmpOrNil < 0 { + ctx.DocumentMatchPool.Put(s.currShould) // advance should searcher to our candidate entry - s.currShould, err = s.shouldSearcher.Advance(s.currentID) + s.currShould, err = s.shouldSearcher.Advance(ctx, s.currentID) if err != nil { return nil, err } - if s.currShould != nil && s.currShould.ID == s.currentID { + if s.currShould != nil && s.currShould.IndexInternalID.Equals(s.currentID) { // score bonus matches should var cons []*search.DocumentMatch if s.currMust != nil { - cons = []*search.DocumentMatch{ - s.currMust, - s.currShould, - } + cons = s.matches + cons[0] = s.currMust + cons[1] = s.currShould } else { - cons = []*search.DocumentMatch{ - s.currShould, - } + cons = s.matches[0:1] + cons[0] = s.currShould } - rv = s.scorer.Score(cons) - err = s.advanceNextMust() + rv = s.scorer.Score(ctx, cons) + err = s.advanceNextMust(ctx, rv) if err != nil { return nil, err } break } else if s.shouldSearcher.Min() == 0 { // match is OK anyway - rv = s.scorer.Score([]*search.DocumentMatch{s.currMust}) - err = s.advanceNextMust() + cons := s.matches[0:1] + cons[0] = s.currMust + rv = s.scorer.Score(ctx, cons) + err = s.advanceNextMust(ctx, rv) if err != nil { return nil, err } break } - } else if s.currShould != nil && s.currShould.ID == s.currentID { + } else if shouldCmpOrNil == 0 { // score bonus matches should var cons []*search.DocumentMatch if s.currMust != nil { - cons = []*search.DocumentMatch{ - s.currMust, - s.currShould, - } + cons = s.matches + cons[0] = s.currMust + cons[1] = s.currShould } else { - cons = []*search.DocumentMatch{ - s.currShould, - } + cons = s.matches[0:1] + cons[0] = s.currShould } - rv = s.scorer.Score(cons) - err = s.advanceNextMust() + rv = s.scorer.Score(ctx, cons) + err = s.advanceNextMust(ctx, rv) if err != nil { return nil, err } break } else if s.shouldSearcher == nil || s.shouldSearcher.Min() == 0 { // match is OK anyway - rv = s.scorer.Score([]*search.DocumentMatch{s.currMust}) - err = s.advanceNextMust() + cons := s.matches[0:1] + cons[0] = s.currMust + rv = s.scorer.Score(ctx, cons) + err = s.advanceNextMust(ctx, rv) if err != nil { return nil, err } break } - err = s.advanceNextMust() + err = s.advanceNextMust(ctx, nil) if err != nil { return nil, err } @@ -255,10 +282,10 @@ func (s *BooleanSearcher) Next() (*search.DocumentMatch, error) { return rv, nil } -func (s *BooleanSearcher) Advance(ID string) (*search.DocumentMatch, error) { +func (s *BooleanSearcher) Advance(ctx *search.SearchContext, ID index.IndexInternalID) (*search.DocumentMatch, error) { if !s.initialized { - err := s.initSearchers() + err := s.initSearchers(ctx) if err != nil { return nil, err } @@ -266,33 +293,42 @@ func (s *BooleanSearcher) Advance(ID string) (*search.DocumentMatch, error) { var err error if s.mustSearcher != nil { - s.currMust, err = s.mustSearcher.Advance(ID) + if s.currMust != nil { + ctx.DocumentMatchPool.Put(s.currMust) + } + s.currMust, err = s.mustSearcher.Advance(ctx, ID) if err != nil { return nil, err } } if s.shouldSearcher != nil { - s.currShould, err = s.shouldSearcher.Advance(ID) + if s.currShould != nil { + ctx.DocumentMatchPool.Put(s.currShould) + } + s.currShould, err = s.shouldSearcher.Advance(ctx, ID) if err != nil { return nil, err } } if s.mustNotSearcher != nil { - s.currMustNot, err = s.mustNotSearcher.Advance(ID) + if s.currMustNot != nil { + ctx.DocumentMatchPool.Put(s.currMustNot) + } + s.currMustNot, err = s.mustNotSearcher.Advance(ctx, ID) if err != nil { return nil, err } } if s.mustSearcher != nil && s.currMust != nil { - s.currentID = s.currMust.ID + s.currentID = s.currMust.IndexInternalID } else if s.mustSearcher == nil && s.currShould != nil { - s.currentID = s.currShould.ID + s.currentID = s.currShould.IndexInternalID } else { - s.currentID = "" + s.currentID = nil } - return s.Next() + return s.Next(ctx) } func (s *BooleanSearcher) Count() uint64 { @@ -333,3 +369,17 @@ func (s *BooleanSearcher) Close() error { func (s *BooleanSearcher) Min() int { return 0 } + +func (s *BooleanSearcher) DocumentMatchPoolSize() int { + rv := 3 + if s.mustSearcher != nil { + rv += s.mustSearcher.DocumentMatchPoolSize() + } + if s.shouldSearcher != nil { + rv += s.shouldSearcher.DocumentMatchPoolSize() + } + if s.mustNotSearcher != nil { + rv += s.mustNotSearcher.DocumentMatchPoolSize() + } + return rv +} diff --git a/vendor/github.com/blevesearch/bleve/search/searchers/search_conjunction.go b/vendor/github.com/blevesearch/bleve/search/searchers/search_conjunction.go index 77c4b1f..c9fe0f8 100644 --- a/vendor/github.com/blevesearch/bleve/search/searchers/search_conjunction.go +++ b/vendor/github.com/blevesearch/bleve/search/searchers/search_conjunction.go @@ -19,14 +19,14 @@ import ( ) type ConjunctionSearcher struct { - initialized bool indexReader index.IndexReader searchers OrderedSearcherList - explain bool queryNorm float64 currs []*search.DocumentMatch - currentID string + maxIDIdx int scorer *scorers.ConjunctionQueryScorer + initialized bool + explain bool } func NewConjunctionSearcher(indexReader index.IndexReader, qsearchers []search.Searcher, explain bool) (*ConjunctionSearcher, error) { @@ -63,24 +63,18 @@ func (s *ConjunctionSearcher) computeQueryNorm() { } } -func (s *ConjunctionSearcher) initSearchers() error { +func (s *ConjunctionSearcher) initSearchers(ctx *search.SearchContext) error { var err error // get all searchers pointing at their first match for i, termSearcher := range s.searchers { - s.currs[i], err = termSearcher.Next() + if s.currs[i] != nil { + ctx.DocumentMatchPool.Put(s.currs[i]) + } + s.currs[i], err = termSearcher.Next(ctx) if err != nil { return err } } - - if len(s.currs) > 0 { - if s.currs[0] != nil { - s.currentID = s.currs[0].ID - } else { - s.currentID = "" - } - } - s.initialized = true return nil } @@ -99,9 +93,9 @@ func (s *ConjunctionSearcher) SetQueryNorm(qnorm float64) { } } -func (s *ConjunctionSearcher) Next() (*search.DocumentMatch, error) { +func (s *ConjunctionSearcher) Next(ctx *search.SearchContext) (*search.DocumentMatch, error) { if !s.initialized { - err := s.initSearchers() + err := s.initSearchers(ctx) if err != nil { return nil, err } @@ -109,68 +103,96 @@ func (s *ConjunctionSearcher) Next() (*search.DocumentMatch, error) { var rv *search.DocumentMatch var err error OUTER: - for s.currentID != "" { - for i, termSearcher := range s.searchers { - if s.currs[i] != nil && s.currs[i].ID != s.currentID { - if s.currentID < s.currs[i].ID { - s.currentID = s.currs[i].ID - continue OUTER + for s.currs[s.maxIDIdx] != nil { + maxID := s.currs[s.maxIDIdx].IndexInternalID + + i := 0 + for i < len(s.currs) { + if s.currs[i] == nil { + return nil, nil + } + + if i == s.maxIDIdx { + i++ + continue + } + + cmp := maxID.Compare(s.currs[i].IndexInternalID) + if cmp == 0 { + i++ + continue + } + + if cmp < 0 { + // maxID < currs[i], so we found a new maxIDIdx + s.maxIDIdx = i + + // advance the positions where [0 <= x < i], since we + // know they were equal to the former max entry + maxID = s.currs[s.maxIDIdx].IndexInternalID + for x := 0; x < i; x++ { + err = s.advanceChild(ctx, x, maxID) + if err != nil { + return nil, err + } } - // this reader doesn't have the currentID, try to advance - s.currs[i], err = termSearcher.Advance(s.currentID) - if err != nil { - return nil, err - } - if s.currs[i] == nil { - s.currentID = "" - continue OUTER - } - if s.currs[i].ID != s.currentID { - // we just advanced, so it doesn't match, it must be greater - // no need to call next - s.currentID = s.currs[i].ID - continue OUTER - } - } else if s.currs[i] == nil { - s.currentID = "" + continue OUTER } - } - // if we get here, a doc matched all readers, sum the score and add it - rv = s.scorer.Score(s.currs) - // prepare for next entry - s.currs[0], err = s.searchers[0].Next() - if err != nil { - return nil, err + // maxID > currs[i], so need to advance searchers[i] + err = s.advanceChild(ctx, i, maxID) + if err != nil { + return nil, err + } + + // don't bump i, so that we'll examine the just-advanced + // currs[i] again } - if s.currs[0] == nil { - s.currentID = "" - } else { - s.currentID = s.currs[0].ID + + // if we get here, a doc matched all readers, so score and add it + rv = s.scorer.Score(ctx, s.currs) + + // we know all the searchers are pointing at the same thing + // so they all need to be bumped + for i, termSearcher := range s.searchers { + if s.currs[i] != rv { + ctx.DocumentMatchPool.Put(s.currs[i]) + } + s.currs[i], err = termSearcher.Next(ctx) + if err != nil { + return nil, err + } } + // don't continue now, wait for the next call to Next() break } return rv, nil } -func (s *ConjunctionSearcher) Advance(ID string) (*search.DocumentMatch, error) { +func (s *ConjunctionSearcher) Advance(ctx *search.SearchContext, ID index.IndexInternalID) (*search.DocumentMatch, error) { if !s.initialized { - err := s.initSearchers() + err := s.initSearchers(ctx) if err != nil { return nil, err } } - var err error - for i, searcher := range s.searchers { - s.currs[i], err = searcher.Advance(ID) + for i := range s.searchers { + err := s.advanceChild(ctx, i, ID) if err != nil { return nil, err } } - s.currentID = ID - return s.Next() + return s.Next(ctx) +} + +func (s *ConjunctionSearcher) advanceChild(ctx *search.SearchContext, i int, ID index.IndexInternalID) (err error) { + if s.currs[i] != nil { + ctx.DocumentMatchPool.Put(s.currs[i]) + } + s.currs[i], err = s.searchers[i].Advance(ctx, ID) + return err } func (s *ConjunctionSearcher) Count() uint64 { @@ -195,3 +217,11 @@ func (s *ConjunctionSearcher) Close() error { func (s *ConjunctionSearcher) Min() int { return 0 } + +func (s *ConjunctionSearcher) DocumentMatchPoolSize() int { + rv := len(s.currs) + for _, s := range s.searchers { + rv += s.DocumentMatchPoolSize() + } + return rv +} diff --git a/vendor/github.com/blevesearch/bleve/search/searchers/search_disjunction.go b/vendor/github.com/blevesearch/bleve/search/searchers/search_disjunction.go index 4686706..daa9f33 100644 --- a/vendor/github.com/blevesearch/bleve/search/searchers/search_disjunction.go +++ b/vendor/github.com/blevesearch/bleve/search/searchers/search_disjunction.go @@ -25,19 +25,31 @@ import ( var DisjunctionMaxClauseCount = 0 type DisjunctionSearcher struct { - initialized bool - indexReader index.IndexReader - searchers OrderedSearcherList - queryNorm float64 - currs []*search.DocumentMatch - currentID string - scorer *scorers.DisjunctionQueryScorer - min float64 + indexReader index.IndexReader + searchers OrderedSearcherList + queryNorm float64 + currs []*search.DocumentMatch + scorer *scorers.DisjunctionQueryScorer + min int + matching []*search.DocumentMatch + matchingIdxs []int + initialized bool +} + +func tooManyClauses(count int) bool { + if DisjunctionMaxClauseCount != 0 && count > DisjunctionMaxClauseCount { + return true + } + return false +} + +func tooManyClausesErr() error { + return fmt.Errorf("TooManyClauses[maxClauseCount is set to %d]", DisjunctionMaxClauseCount) } func NewDisjunctionSearcher(indexReader index.IndexReader, qsearchers []search.Searcher, min float64, explain bool) (*DisjunctionSearcher, error) { - if DisjunctionMaxClauseCount != 0 && len(qsearchers) > DisjunctionMaxClauseCount { - return nil, fmt.Errorf("TooManyClauses[maxClauseCount is set to %d]", DisjunctionMaxClauseCount) + if tooManyClauses(len(qsearchers)) { + return nil, tooManyClausesErr() } // build the downstream searchers searchers := make(OrderedSearcherList, len(qsearchers)) @@ -48,11 +60,13 @@ func NewDisjunctionSearcher(indexReader index.IndexReader, qsearchers []search.S sort.Sort(sort.Reverse(searchers)) // build our searcher rv := DisjunctionSearcher{ - indexReader: indexReader, - searchers: searchers, - currs: make([]*search.DocumentMatch, len(searchers)), - scorer: scorers.NewDisjunctionQueryScorer(explain), - min: min, + indexReader: indexReader, + searchers: searchers, + currs: make([]*search.DocumentMatch, len(searchers)), + scorer: scorers.NewDisjunctionQueryScorer(explain), + min: int(min), + matching: make([]*search.DocumentMatch, len(searchers)), + matchingIdxs: make([]int, len(searchers)), } rv.computeQueryNorm() return &rv, nil @@ -72,29 +86,51 @@ func (s *DisjunctionSearcher) computeQueryNorm() { } } -func (s *DisjunctionSearcher) initSearchers() error { +func (s *DisjunctionSearcher) initSearchers(ctx *search.SearchContext) error { var err error // get all searchers pointing at their first match for i, termSearcher := range s.searchers { - s.currs[i], err = termSearcher.Next() + if s.currs[i] != nil { + ctx.DocumentMatchPool.Put(s.currs[i]) + } + s.currs[i], err = termSearcher.Next(ctx) if err != nil { return err } } - s.currentID = s.nextSmallestID() + s.updateMatches() s.initialized = true return nil } -func (s *DisjunctionSearcher) nextSmallestID() string { - rv := "" - for _, curr := range s.currs { - if curr != nil && (curr.ID < rv || rv == "") { - rv = curr.ID +func (s *DisjunctionSearcher) updateMatches() { + matching := s.matching[:0] + matchingIdxs := s.matchingIdxs[:0] + + for i, curr := range s.currs { + if curr == nil { + continue } + + if len(matching) > 0 { + cmp := curr.IndexInternalID.Compare(matching[0].IndexInternalID) + if cmp > 0 { + continue + } + + if cmp < 0 { + matching = matching[:0] + matchingIdxs = matchingIdxs[:0] + } + } + + matching = append(matching, curr) + matchingIdxs = append(matchingIdxs, i) } - return rv + + s.matching = matching + s.matchingIdxs = matchingIdxs } func (s *DisjunctionSearcher) Weight() float64 { @@ -111,51 +147,44 @@ func (s *DisjunctionSearcher) SetQueryNorm(qnorm float64) { } } -func (s *DisjunctionSearcher) Next() (*search.DocumentMatch, error) { +func (s *DisjunctionSearcher) Next(ctx *search.SearchContext) (*search.DocumentMatch, error) { if !s.initialized { - err := s.initSearchers() + err := s.initSearchers(ctx) if err != nil { return nil, err } } var err error var rv *search.DocumentMatch - matching := make([]*search.DocumentMatch, 0, len(s.searchers)) found := false - for !found && s.currentID != "" { - for _, curr := range s.currs { - if curr != nil && curr.ID == s.currentID { - matching = append(matching, curr) - } - } - - if len(matching) >= int(s.min) { + for !found && len(s.matching) > 0 { + if len(s.matching) >= s.min { found = true // score this match - rv = s.scorer.Score(matching, len(matching), len(s.searchers)) + rv = s.scorer.Score(ctx, s.matching, len(s.matching), len(s.searchers)) } - // reset matching - matching = make([]*search.DocumentMatch, 0) // invoke next on all the matching searchers - for i, curr := range s.currs { - if curr != nil && curr.ID == s.currentID { - searcher := s.searchers[i] - s.currs[i], err = searcher.Next() - if err != nil { - return nil, err - } + for _, i := range s.matchingIdxs { + searcher := s.searchers[i] + if s.currs[i] != rv { + ctx.DocumentMatchPool.Put(s.currs[i]) + } + s.currs[i], err = searcher.Next(ctx) + if err != nil { + return nil, err } } - s.currentID = s.nextSmallestID() + + s.updateMatches() } return rv, nil } -func (s *DisjunctionSearcher) Advance(ID string) (*search.DocumentMatch, error) { +func (s *DisjunctionSearcher) Advance(ctx *search.SearchContext, ID index.IndexInternalID) (*search.DocumentMatch, error) { if !s.initialized { - err := s.initSearchers() + err := s.initSearchers(ctx) if err != nil { return nil, err } @@ -163,15 +192,18 @@ func (s *DisjunctionSearcher) Advance(ID string) (*search.DocumentMatch, error) // get all searchers pointing at their first match var err error for i, termSearcher := range s.searchers { - s.currs[i], err = termSearcher.Advance(ID) + if s.currs[i] != nil { + ctx.DocumentMatchPool.Put(s.currs[i]) + } + s.currs[i], err = termSearcher.Advance(ctx, ID) if err != nil { return nil, err } } - s.currentID = s.nextSmallestID() + s.updateMatches() - return s.Next() + return s.Next(ctx) } func (s *DisjunctionSearcher) Count() uint64 { @@ -194,5 +226,13 @@ func (s *DisjunctionSearcher) Close() error { } func (s *DisjunctionSearcher) Min() int { - return int(s.min) // FIXME just make this an int + return s.min +} + +func (s *DisjunctionSearcher) DocumentMatchPoolSize() int { + rv := len(s.currs) + for _, s := range s.searchers { + rv += s.DocumentMatchPoolSize() + } + return rv } diff --git a/vendor/github.com/blevesearch/bleve/search/searchers/search_docid.go b/vendor/github.com/blevesearch/bleve/search/searchers/search_docid.go index 8d0f6cc..33b9b4c 100644 --- a/vendor/github.com/blevesearch/bleve/search/searchers/search_docid.go +++ b/vendor/github.com/blevesearch/bleve/search/searchers/search_docid.go @@ -10,8 +10,6 @@ package searchers import ( - "sort" - "github.com/blevesearch/bleve/index" "github.com/blevesearch/bleve/search" "github.com/blevesearch/bleve/search/scorers" @@ -19,54 +17,28 @@ import ( // DocIDSearcher returns documents matching a predefined set of identifiers. type DocIDSearcher struct { - ids []string - current int - scorer *scorers.ConstantScorer + reader index.DocIDReader + scorer *scorers.ConstantScorer + count int } func NewDocIDSearcher(indexReader index.IndexReader, ids []string, boost float64, explain bool) (searcher *DocIDSearcher, err error) { - kept := make([]string, len(ids)) - copy(kept, ids) - sort.Strings(kept) - - if len(ids) > 0 { - var idReader index.DocIDReader - endTerm := string(incrementBytes([]byte(kept[len(kept)-1]))) - idReader, err = indexReader.DocIDReader(kept[0], endTerm) - if err != nil { - return nil, err - } - defer func() { - if cerr := idReader.Close(); err == nil && cerr != nil { - err = cerr - } - }() - j := 0 - for _, id := range kept { - doc, err := idReader.Advance(id) - if err != nil { - return nil, err - } - // Non-duplicate match - if doc == id && (j == 0 || kept[j-1] != id) { - kept[j] = id - j++ - } - } - kept = kept[:j] + reader, err := indexReader.DocIDReaderOnly(ids) + if err != nil { + return nil, err } - scorer := scorers.NewConstantScorer(1.0, boost, explain) return &DocIDSearcher{ - ids: kept, scorer: scorer, + reader: reader, + count: len(ids), }, nil } func (s *DocIDSearcher) Count() uint64 { - return uint64(len(s.ids)) + return uint64(s.count) } func (s *DocIDSearcher) Weight() float64 { @@ -77,20 +49,30 @@ func (s *DocIDSearcher) SetQueryNorm(qnorm float64) { s.scorer.SetQueryNorm(qnorm) } -func (s *DocIDSearcher) Next() (*search.DocumentMatch, error) { - if s.current >= len(s.ids) { +func (s *DocIDSearcher) Next(ctx *search.SearchContext) (*search.DocumentMatch, error) { + docidMatch, err := s.reader.Next() + if err != nil { + return nil, err + } + if docidMatch == nil { return nil, nil } - id := s.ids[s.current] - s.current++ - docMatch := s.scorer.Score(id) - return docMatch, nil + docMatch := s.scorer.Score(ctx, docidMatch) + return docMatch, nil } -func (s *DocIDSearcher) Advance(ID string) (*search.DocumentMatch, error) { - s.current = sort.SearchStrings(s.ids, ID) - return s.Next() +func (s *DocIDSearcher) Advance(ctx *search.SearchContext, ID index.IndexInternalID) (*search.DocumentMatch, error) { + docidMatch, err := s.reader.Advance(ID) + if err != nil { + return nil, err + } + if docidMatch == nil { + return nil, nil + } + + docMatch := s.scorer.Score(ctx, docidMatch) + return docMatch, nil } func (s *DocIDSearcher) Close() error { @@ -100,3 +82,7 @@ func (s *DocIDSearcher) Close() error { func (s *DocIDSearcher) Min() int { return 0 } + +func (s *DocIDSearcher) DocumentMatchPoolSize() int { + return 1 +} diff --git a/vendor/github.com/blevesearch/bleve/search/searchers/search_fuzzy.go b/vendor/github.com/blevesearch/bleve/search/searchers/search_fuzzy.go index 766a6bb..2cd44eb 100644 --- a/vendor/github.com/blevesearch/bleve/search/searchers/search_fuzzy.go +++ b/vendor/github.com/blevesearch/bleve/search/searchers/search_fuzzy.go @@ -32,36 +32,13 @@ func NewFuzzySearcher(indexReader index.IndexReader, term string, prefix, fuzzin } } - // find the terms with this prefix - var fieldDict index.FieldDict - var err error - if len(prefixTerm) > 0 { - fieldDict, err = indexReader.FieldDictPrefix(field, []byte(prefixTerm)) - } else { - fieldDict, err = indexReader.FieldDict(field) - } - - // enumerate terms and check levenshtein distance - candidateTerms := make([]string, 0) - tfd, err := fieldDict.Next() - for err == nil && tfd != nil { - ld, exceeded := search.LevenshteinDistanceMax(&term, &tfd.Term, fuzziness) - if !exceeded && ld <= fuzziness { - candidateTerms = append(candidateTerms, tfd.Term) - } - tfd, err = fieldDict.Next() - } - if err != nil { - return nil, err - } - - err = fieldDict.Close() + candidateTerms, err := findFuzzyCandidateTerms(indexReader, &term, fuzziness, field, prefixTerm) if err != nil { return nil, err } // enumerate all the terms in the range - qsearchers := make([]search.Searcher, 0, 25) + qsearchers := make([]search.Searcher, 0, len(candidateTerms)) for _, cterm := range candidateTerms { qsearcher, err := NewTermSearcher(indexReader, cterm, field, boost, explain) @@ -87,6 +64,37 @@ func NewFuzzySearcher(indexReader index.IndexReader, term string, prefix, fuzzin searcher: searcher, }, nil } + +func findFuzzyCandidateTerms(indexReader index.IndexReader, term *string, fuzziness int, field, prefixTerm string) (rv []string, err error) { + rv = make([]string, 0) + var fieldDict index.FieldDict + if len(prefixTerm) > 0 { + fieldDict, err = indexReader.FieldDictPrefix(field, []byte(prefixTerm)) + } else { + fieldDict, err = indexReader.FieldDict(field) + } + defer func() { + if cerr := fieldDict.Close(); cerr != nil && err == nil { + err = cerr + } + }() + + // enumerate terms and check levenshtein distance + tfd, err := fieldDict.Next() + for err == nil && tfd != nil { + ld, exceeded := search.LevenshteinDistanceMax(term, &tfd.Term, fuzziness) + if !exceeded && ld <= fuzziness { + rv = append(rv, tfd.Term) + if tooManyClauses(len(rv)) { + return rv, tooManyClausesErr() + } + } + tfd, err = fieldDict.Next() + } + + return rv, err +} + func (s *FuzzySearcher) Count() uint64 { return s.searcher.Count() } @@ -99,13 +107,13 @@ func (s *FuzzySearcher) SetQueryNorm(qnorm float64) { s.searcher.SetQueryNorm(qnorm) } -func (s *FuzzySearcher) Next() (*search.DocumentMatch, error) { - return s.searcher.Next() +func (s *FuzzySearcher) Next(ctx *search.SearchContext) (*search.DocumentMatch, error) { + return s.searcher.Next(ctx) } -func (s *FuzzySearcher) Advance(ID string) (*search.DocumentMatch, error) { - return s.searcher.Next() +func (s *FuzzySearcher) Advance(ctx *search.SearchContext, ID index.IndexInternalID) (*search.DocumentMatch, error) { + return s.searcher.Advance(ctx, ID) } func (s *FuzzySearcher) Close() error { @@ -115,3 +123,7 @@ func (s *FuzzySearcher) Close() error { func (s *FuzzySearcher) Min() int { return 0 } + +func (s *FuzzySearcher) DocumentMatchPoolSize() int { + return s.searcher.DocumentMatchPoolSize() +} diff --git a/vendor/github.com/blevesearch/bleve/search/searchers/search_match_all.go b/vendor/github.com/blevesearch/bleve/search/searchers/search_match_all.go index 657bbaf..5659f88 100644 --- a/vendor/github.com/blevesearch/bleve/search/searchers/search_match_all.go +++ b/vendor/github.com/blevesearch/bleve/search/searchers/search_match_all.go @@ -19,10 +19,15 @@ type MatchAllSearcher struct { indexReader index.IndexReader reader index.DocIDReader scorer *scorers.ConstantScorer + count uint64 } func NewMatchAllSearcher(indexReader index.IndexReader, boost float64, explain bool) (*MatchAllSearcher, error) { - reader, err := indexReader.DocIDReader("", "") + reader, err := indexReader.DocIDReaderAll() + if err != nil { + return nil, err + } + count, err := indexReader.DocCount() if err != nil { return nil, err } @@ -31,11 +36,12 @@ func NewMatchAllSearcher(indexReader index.IndexReader, boost float64, explain b indexReader: indexReader, reader: reader, scorer: scorer, + count: count, }, nil } func (s *MatchAllSearcher) Count() uint64 { - return s.indexReader.DocCount() + return s.count } func (s *MatchAllSearcher) Weight() float64 { @@ -46,35 +52,35 @@ func (s *MatchAllSearcher) SetQueryNorm(qnorm float64) { s.scorer.SetQueryNorm(qnorm) } -func (s *MatchAllSearcher) Next() (*search.DocumentMatch, error) { +func (s *MatchAllSearcher) Next(ctx *search.SearchContext) (*search.DocumentMatch, error) { id, err := s.reader.Next() if err != nil { return nil, err } - if id == "" { + if id == nil { return nil, nil } // score match - docMatch := s.scorer.Score(id) + docMatch := s.scorer.Score(ctx, id) // return doc match return docMatch, nil } -func (s *MatchAllSearcher) Advance(ID string) (*search.DocumentMatch, error) { +func (s *MatchAllSearcher) Advance(ctx *search.SearchContext, ID index.IndexInternalID) (*search.DocumentMatch, error) { id, err := s.reader.Advance(ID) if err != nil { return nil, err } - if id == "" { + if id == nil { return nil, nil } // score match - docMatch := s.scorer.Score(id) + docMatch := s.scorer.Score(ctx, id) // return doc match return docMatch, nil @@ -87,3 +93,7 @@ func (s *MatchAllSearcher) Close() error { func (s *MatchAllSearcher) Min() int { return 0 } + +func (s *MatchAllSearcher) DocumentMatchPoolSize() int { + return 1 +} diff --git a/vendor/github.com/blevesearch/bleve/search/searchers/search_match_none.go b/vendor/github.com/blevesearch/bleve/search/searchers/search_match_none.go index 87881f5..c3159dc 100644 --- a/vendor/github.com/blevesearch/bleve/search/searchers/search_match_none.go +++ b/vendor/github.com/blevesearch/bleve/search/searchers/search_match_none.go @@ -36,11 +36,11 @@ func (s *MatchNoneSearcher) SetQueryNorm(qnorm float64) { } -func (s *MatchNoneSearcher) Next() (*search.DocumentMatch, error) { +func (s *MatchNoneSearcher) Next(ctx *search.SearchContext) (*search.DocumentMatch, error) { return nil, nil } -func (s *MatchNoneSearcher) Advance(ID string) (*search.DocumentMatch, error) { +func (s *MatchNoneSearcher) Advance(ctx *search.SearchContext, ID index.IndexInternalID) (*search.DocumentMatch, error) { return nil, nil } @@ -51,3 +51,7 @@ func (s *MatchNoneSearcher) Close() error { func (s *MatchNoneSearcher) Min() int { return 0 } + +func (s *MatchNoneSearcher) DocumentMatchPoolSize() int { + return 0 +} diff --git a/vendor/github.com/blevesearch/bleve/search/searchers/search_numeric_range.go b/vendor/github.com/blevesearch/bleve/search/searchers/search_numeric_range.go index 92134af..03cd495 100644 --- a/vendor/github.com/blevesearch/bleve/search/searchers/search_numeric_range.go +++ b/vendor/github.com/blevesearch/bleve/search/searchers/search_numeric_range.go @@ -57,6 +57,9 @@ func NewNumericRangeSearcher(indexReader index.IndexReader, min *float64, max *f // FIXME hard-coded precision, should match field declaration termRanges := splitInt64Range(minInt64, maxInt64, 4) terms := termRanges.Enumerate() + if tooManyClauses(len(terms)) { + return nil, tooManyClausesErr() + } // enumerate all the terms in the range qsearchers := make([]search.Searcher, len(terms)) for i, term := range terms { @@ -93,12 +96,12 @@ func (s *NumericRangeSearcher) SetQueryNorm(qnorm float64) { s.searcher.SetQueryNorm(qnorm) } -func (s *NumericRangeSearcher) Next() (*search.DocumentMatch, error) { - return s.searcher.Next() +func (s *NumericRangeSearcher) Next(ctx *search.SearchContext) (*search.DocumentMatch, error) { + return s.searcher.Next(ctx) } -func (s *NumericRangeSearcher) Advance(ID string) (*search.DocumentMatch, error) { - return s.searcher.Advance(ID) +func (s *NumericRangeSearcher) Advance(ctx *search.SearchContext, ID index.IndexInternalID) (*search.DocumentMatch, error) { + return s.searcher.Advance(ctx, ID) } func (s *NumericRangeSearcher) Close() error { @@ -212,3 +215,7 @@ func newRangeBytes(minBytes, maxBytes []byte) *termRange { func (s *NumericRangeSearcher) Min() int { return 0 } + +func (s *NumericRangeSearcher) DocumentMatchPoolSize() int { + return s.searcher.DocumentMatchPoolSize() +} diff --git a/vendor/github.com/blevesearch/bleve/search/searchers/search_phrase.go b/vendor/github.com/blevesearch/bleve/search/searchers/search_phrase.go index fd318b4..6456420 100644 --- a/vendor/github.com/blevesearch/bleve/search/searchers/search_phrase.go +++ b/vendor/github.com/blevesearch/bleve/search/searchers/search_phrase.go @@ -52,11 +52,11 @@ func (s *PhraseSearcher) computeQueryNorm() { } } -func (s *PhraseSearcher) initSearchers() error { +func (s *PhraseSearcher) initSearchers(ctx *search.SearchContext) error { var err error // get all searchers pointing at their first match if s.mustSearcher != nil { - s.currMust, err = s.mustSearcher.Next() + s.currMust, err = s.mustSearcher.Next(ctx) if err != nil { return err } @@ -66,11 +66,11 @@ func (s *PhraseSearcher) initSearchers() error { return nil } -func (s *PhraseSearcher) advanceNextMust() error { +func (s *PhraseSearcher) advanceNextMust(ctx *search.SearchContext) error { var err error if s.mustSearcher != nil { - s.currMust, err = s.mustSearcher.Next() + s.currMust, err = s.mustSearcher.Next(ctx) if err != nil { return err } @@ -90,9 +90,9 @@ func (s *PhraseSearcher) SetQueryNorm(qnorm float64) { s.mustSearcher.SetQueryNorm(qnorm) } -func (s *PhraseSearcher) Next() (*search.DocumentMatch, error) { +func (s *PhraseSearcher) Next(ctx *search.SearchContext) (*search.DocumentMatch, error) { if !s.initialized { - err := s.initSearchers() + err := s.initSearchers(ctx) if err != nil { return nil, err } @@ -144,14 +144,14 @@ func (s *PhraseSearcher) Next() (*search.DocumentMatch, error) { // return match rv = s.currMust rv.Locations = rvftlm - err := s.advanceNextMust() + err := s.advanceNextMust(ctx) if err != nil { return nil, err } return rv, nil } - err := s.advanceNextMust() + err := s.advanceNextMust(ctx) if err != nil { return nil, err } @@ -160,19 +160,19 @@ func (s *PhraseSearcher) Next() (*search.DocumentMatch, error) { return nil, nil } -func (s *PhraseSearcher) Advance(ID string) (*search.DocumentMatch, error) { +func (s *PhraseSearcher) Advance(ctx *search.SearchContext, ID index.IndexInternalID) (*search.DocumentMatch, error) { if !s.initialized { - err := s.initSearchers() + err := s.initSearchers(ctx) if err != nil { return nil, err } } var err error - s.currMust, err = s.mustSearcher.Advance(ID) + s.currMust, err = s.mustSearcher.Advance(ctx, ID) if err != nil { return nil, err } - return s.Next() + return s.Next(ctx) } func (s *PhraseSearcher) Count() uint64 { @@ -195,3 +195,7 @@ func (s *PhraseSearcher) Close() error { func (s *PhraseSearcher) Min() int { return 0 } + +func (s *PhraseSearcher) DocumentMatchPoolSize() int { + return s.mustSearcher.DocumentMatchPoolSize() + 1 +} diff --git a/vendor/github.com/blevesearch/bleve/search/searchers/search_regexp.go b/vendor/github.com/blevesearch/bleve/search/searchers/search_regexp.go index fe994f5..3ce61d8 100644 --- a/vendor/github.com/blevesearch/bleve/search/searchers/search_regexp.go +++ b/vendor/github.com/blevesearch/bleve/search/searchers/search_regexp.go @@ -27,39 +27,20 @@ type RegexpSearcher struct { func NewRegexpSearcher(indexReader index.IndexReader, pattern *regexp.Regexp, field string, boost float64, explain bool) (*RegexpSearcher, error) { prefixTerm, complete := pattern.LiteralPrefix() - candidateTerms := make([]string, 0) + var candidateTerms []string if complete { // there is no pattern - candidateTerms = append(candidateTerms, prefixTerm) + candidateTerms = []string{prefixTerm} } else { - var fieldDict index.FieldDict var err error - if len(prefixTerm) > 0 { - fieldDict, err = indexReader.FieldDictPrefix(field, []byte(prefixTerm)) - } else { - fieldDict, err = indexReader.FieldDict(field) - } - - // enumerate the terms and check against regexp - tfd, err := fieldDict.Next() - for err == nil && tfd != nil { - if pattern.MatchString(tfd.Term) { - candidateTerms = append(candidateTerms, tfd.Term) - } - tfd, err = fieldDict.Next() - } - if err != nil { - return nil, err - } - - err = fieldDict.Close() + candidateTerms, err = findRegexpCandidateTerms(indexReader, pattern, field, prefixTerm) if err != nil { return nil, err } } // enumerate all the terms in the range - qsearchers := make([]search.Searcher, 0, 25) + qsearchers := make([]search.Searcher, 0, len(candidateTerms)) for _, cterm := range candidateTerms { qsearcher, err := NewTermSearcher(indexReader, cterm, field, boost, explain) @@ -83,6 +64,36 @@ func NewRegexpSearcher(indexReader index.IndexReader, pattern *regexp.Regexp, fi searcher: searcher, }, nil } + +func findRegexpCandidateTerms(indexReader index.IndexReader, pattern *regexp.Regexp, field, prefixTerm string) (rv []string, err error) { + rv = make([]string, 0) + var fieldDict index.FieldDict + if len(prefixTerm) > 0 { + fieldDict, err = indexReader.FieldDictPrefix(field, []byte(prefixTerm)) + } else { + fieldDict, err = indexReader.FieldDict(field) + } + defer func() { + if cerr := fieldDict.Close(); cerr != nil && err == nil { + err = cerr + } + }() + + // enumerate the terms and check against regexp + tfd, err := fieldDict.Next() + for err == nil && tfd != nil { + if pattern.MatchString(tfd.Term) { + rv = append(rv, tfd.Term) + if tooManyClauses(len(rv)) { + return rv, tooManyClausesErr() + } + } + tfd, err = fieldDict.Next() + } + + return rv, err +} + func (s *RegexpSearcher) Count() uint64 { return s.searcher.Count() } @@ -95,13 +106,13 @@ func (s *RegexpSearcher) SetQueryNorm(qnorm float64) { s.searcher.SetQueryNorm(qnorm) } -func (s *RegexpSearcher) Next() (*search.DocumentMatch, error) { - return s.searcher.Next() +func (s *RegexpSearcher) Next(ctx *search.SearchContext) (*search.DocumentMatch, error) { + return s.searcher.Next(ctx) } -func (s *RegexpSearcher) Advance(ID string) (*search.DocumentMatch, error) { - return s.searcher.Next() +func (s *RegexpSearcher) Advance(ctx *search.SearchContext, ID index.IndexInternalID) (*search.DocumentMatch, error) { + return s.searcher.Advance(ctx, ID) } func (s *RegexpSearcher) Close() error { @@ -111,3 +122,7 @@ func (s *RegexpSearcher) Close() error { func (s *RegexpSearcher) Min() int { return 0 } + +func (s *RegexpSearcher) DocumentMatchPoolSize() int { + return s.searcher.DocumentMatchPoolSize() +} diff --git a/vendor/github.com/blevesearch/bleve/search/searchers/search_term.go b/vendor/github.com/blevesearch/bleve/search/searchers/search_term.go index 2ea3a9d..fcdf6cc 100644 --- a/vendor/github.com/blevesearch/bleve/search/searchers/search_term.go +++ b/vendor/github.com/blevesearch/bleve/search/searchers/search_term.go @@ -19,17 +19,22 @@ type TermSearcher struct { indexReader index.IndexReader term string field string - explain bool reader index.TermFieldReader scorer *scorers.TermQueryScorer + tfd index.TermFieldDoc + explain bool } func NewTermSearcher(indexReader index.IndexReader, term string, field string, boost float64, explain bool) (*TermSearcher, error) { - reader, err := indexReader.TermFieldReader([]byte(term), field) + reader, err := indexReader.TermFieldReader([]byte(term), field, true, true, true) if err != nil { return nil, err } - scorer := scorers.NewTermQueryScorer(term, field, boost, indexReader.DocCount(), reader.Count(), explain) + count, err := indexReader.DocCount() + if err != nil { + return nil, err + } + scorer := scorers.NewTermQueryScorer(term, field, boost, count, reader.Count(), explain) return &TermSearcher{ indexReader: indexReader, term: term, @@ -52,8 +57,8 @@ func (s *TermSearcher) SetQueryNorm(qnorm float64) { s.scorer.SetQueryNorm(qnorm) } -func (s *TermSearcher) Next() (*search.DocumentMatch, error) { - termMatch, err := s.reader.Next() +func (s *TermSearcher) Next(ctx *search.SearchContext) (*search.DocumentMatch, error) { + termMatch, err := s.reader.Next(s.tfd.Reset()) if err != nil { return nil, err } @@ -63,14 +68,14 @@ func (s *TermSearcher) Next() (*search.DocumentMatch, error) { } // score match - docMatch := s.scorer.Score(termMatch) + docMatch := s.scorer.Score(ctx, termMatch) // return doc match return docMatch, nil } -func (s *TermSearcher) Advance(ID string) (*search.DocumentMatch, error) { - termMatch, err := s.reader.Advance(ID) +func (s *TermSearcher) Advance(ctx *search.SearchContext, ID index.IndexInternalID) (*search.DocumentMatch, error) { + termMatch, err := s.reader.Advance(ID, s.tfd.Reset()) if err != nil { return nil, err } @@ -80,7 +85,7 @@ func (s *TermSearcher) Advance(ID string) (*search.DocumentMatch, error) { } // score match - docMatch := s.scorer.Score(termMatch) + docMatch := s.scorer.Score(ctx, termMatch) // return doc match return docMatch, nil @@ -93,3 +98,7 @@ func (s *TermSearcher) Close() error { func (s *TermSearcher) Min() int { return 0 } + +func (s *TermSearcher) DocumentMatchPoolSize() int { + return 1 +} diff --git a/vendor/github.com/blevesearch/bleve/search/searchers/search_term_prefix.go b/vendor/github.com/blevesearch/bleve/search/searchers/search_term_prefix.go index 8fefee0..35caf71 100644 --- a/vendor/github.com/blevesearch/bleve/search/searchers/search_term_prefix.go +++ b/vendor/github.com/blevesearch/bleve/search/searchers/search_term_prefix.go @@ -70,13 +70,13 @@ func (s *TermPrefixSearcher) SetQueryNorm(qnorm float64) { s.searcher.SetQueryNorm(qnorm) } -func (s *TermPrefixSearcher) Next() (*search.DocumentMatch, error) { - return s.searcher.Next() +func (s *TermPrefixSearcher) Next(ctx *search.SearchContext) (*search.DocumentMatch, error) { + return s.searcher.Next(ctx) } -func (s *TermPrefixSearcher) Advance(ID string) (*search.DocumentMatch, error) { - return s.searcher.Next() +func (s *TermPrefixSearcher) Advance(ctx *search.SearchContext, ID index.IndexInternalID) (*search.DocumentMatch, error) { + return s.searcher.Advance(ctx, ID) } func (s *TermPrefixSearcher) Close() error { @@ -86,3 +86,7 @@ func (s *TermPrefixSearcher) Close() error { func (s *TermPrefixSearcher) Min() int { return 0 } + +func (s *TermPrefixSearcher) DocumentMatchPoolSize() int { + return s.searcher.DocumentMatchPoolSize() +} diff --git a/vendor/github.com/blevesearch/bleve/search/sort.go b/vendor/github.com/blevesearch/bleve/search/sort.go new file mode 100644 index 0000000..6428742 --- /dev/null +++ b/vendor/github.com/blevesearch/bleve/search/sort.go @@ -0,0 +1,488 @@ +// Copyright (c) 2014 Couchbase, Inc. +// Licensed under the Apache License, Version 2.0 (the "License"); you may not use this file +// except in compliance with the License. You may obtain a copy of the License at +// http://www.apache.org/licenses/LICENSE-2.0 +// Unless required by applicable law or agreed to in writing, software distributed under the +// License is distributed on an "AS IS" BASIS, WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, +// either express or implied. See the License for the specific language governing permissions +// and limitations under the License. + +package search + +import ( + "encoding/json" + "fmt" + "sort" + "strings" + + "github.com/blevesearch/bleve/numeric_util" +) + +var HighTerm = strings.Repeat(string([]byte{0xff}), 10) +var LowTerm = string([]byte{0x00}) + +type SearchSort interface { + Value(a *DocumentMatch) string + Descending() bool + + RequiresDocID() bool + RequiresScoring() bool + RequiresFields() []string +} + +func ParseSearchSortObj(input map[string]interface{}) (SearchSort, error) { + descending, ok := input["desc"].(bool) + by, ok := input["by"].(string) + if !ok { + return nil, fmt.Errorf("search sort must specify by") + } + switch by { + case "id": + return &SortDocID{ + Desc: descending, + }, nil + case "score": + return &SortScore{ + Desc: descending, + }, nil + case "field": + field, ok := input["field"].(string) + if !ok { + return nil, fmt.Errorf("search sort mode field must specify field") + } + rv := &SortField{ + Field: field, + Desc: descending, + } + typ, ok := input["type"].(string) + if ok { + switch typ { + case "auto": + rv.Type = SortFieldAuto + case "string": + rv.Type = SortFieldAsString + case "number": + rv.Type = SortFieldAsNumber + case "date": + rv.Type = SortFieldAsDate + default: + return nil, fmt.Errorf("unkown sort field type: %s", typ) + } + } + mode, ok := input["mode"].(string) + if ok { + switch mode { + case "default": + rv.Mode = SortFieldDefault + case "min": + rv.Mode = SortFieldMin + case "max": + rv.Mode = SortFieldMax + default: + return nil, fmt.Errorf("unknown sort field mode: %s", mode) + } + } + missing, ok := input["missing"].(string) + if ok { + switch missing { + case "first": + rv.Missing = SortFieldMissingFirst + case "last": + rv.Missing = SortFieldMissingLast + default: + return nil, fmt.Errorf("unknown sort field missing: %s", missing) + } + } + return rv, nil + } + + return nil, fmt.Errorf("unknown search sort by: %s", by) +} + +func ParseSearchSortString(input string) SearchSort { + descending := false + if strings.HasPrefix(input, "-") { + descending = true + input = input[1:] + } else if strings.HasPrefix(input, "+") { + input = input[1:] + } + if input == "_id" { + return &SortDocID{ + Desc: descending, + } + } else if input == "_score" { + return &SortScore{ + Desc: descending, + } + } + return &SortField{ + Field: input, + Desc: descending, + } +} + +func ParseSearchSortJSON(input json.RawMessage) (SearchSort, error) { + // first try to parse it as string + var sortString string + err := json.Unmarshal(input, &sortString) + if err != nil { + var sortObj map[string]interface{} + err = json.Unmarshal(input, &sortObj) + if err != nil { + return nil, err + } + return ParseSearchSortObj(sortObj) + } + return ParseSearchSortString(sortString), nil +} + +func ParseSortOrderStrings(in []string) SortOrder { + rv := make(SortOrder, 0, len(in)) + for _, i := range in { + ss := ParseSearchSortString(i) + rv = append(rv, ss) + } + return rv +} + +func ParseSortOrderJSON(in []json.RawMessage) (SortOrder, error) { + rv := make(SortOrder, 0, len(in)) + for _, i := range in { + ss, err := ParseSearchSortJSON(i) + if err != nil { + return nil, err + } + rv = append(rv, ss) + } + return rv, nil +} + +type SortOrder []SearchSort + +func (so SortOrder) Value(doc *DocumentMatch) { + for _, soi := range so { + doc.Sort = append(doc.Sort, soi.Value(doc)) + } +} + +// Compare will compare two document matches using the specified sort order +// if both are numbers, we avoid converting back to term +func (so SortOrder) Compare(cachedScoring, cachedDesc []bool, i, j *DocumentMatch) int { + // compare the documents on all search sorts until a differences is found + for x := range so { + c := 0 + if cachedScoring[x] { + if i.Score < j.Score { + c = -1 + } else if i.Score > j.Score { + c = 1 + } + } else { + iVal := i.Sort[x] + jVal := j.Sort[x] + c = strings.Compare(iVal, jVal) + } + + if c == 0 { + continue + } + if cachedDesc[x] { + c = -c + } + return c + } + // if they are the same at this point, impose order based on index natural sort order + if i.HitNumber == j.HitNumber { + return 0 + } else if i.HitNumber > j.HitNumber { + return 1 + } + return -1 +} + +func (so SortOrder) RequiresScore() bool { + rv := false + for _, soi := range so { + if soi.RequiresScoring() { + rv = true + } + } + return rv +} + +func (so SortOrder) RequiresDocID() bool { + rv := false + for _, soi := range so { + if soi.RequiresDocID() { + rv = true + } + } + return rv +} + +func (so SortOrder) RequiredFields() []string { + var rv []string + for _, soi := range so { + rv = append(rv, soi.RequiresFields()...) + } + return rv +} + +func (so SortOrder) CacheIsScore() []bool { + var rv []bool + for _, soi := range so { + rv = append(rv, soi.RequiresScoring()) + } + return rv +} + +func (so SortOrder) CacheDescending() []bool { + var rv []bool + for _, soi := range so { + rv = append(rv, soi.Descending()) + } + return rv +} + +// SortFieldType lets you control some internal sort behavior +// normally leaving this to the zero-value of SortFieldAuto is fine +type SortFieldType int + +const ( + // SortFieldAuto applies heuristics attempt to automatically sort correctly + SortFieldAuto SortFieldType = iota + // SortFieldAsString forces sort as string (no prefix coded terms removed) + SortFieldAsString + // SortFieldAsNumber forces sort as string (prefix coded terms with shift > 0 removed) + SortFieldAsNumber + // SortFieldAsDate forces sort as string (prefix coded terms with shift > 0 removed) + SortFieldAsDate +) + +// SortFieldMode describes the behavior if the field has multiple values +type SortFieldMode int + +const ( + // SortFieldDefault uses the first (or only) value, this is the default zero-value + SortFieldDefault SortFieldMode = iota // FIXME name is confusing + // SortFieldMin uses the minimum value + SortFieldMin + // SortFieldMax uses the maximum value + SortFieldMax +) + +// SortFieldMissing controls where documents missing a field value should be sorted +type SortFieldMissing int + +const ( + // SortFieldMissingLast sorts documents missing a field at the end + SortFieldMissingLast SortFieldMissing = iota + + // SortFieldMissingFirst sorts documents missing a field at the beginning + SortFieldMissingFirst +) + +// SortField will sort results by the value of a stored field +// Field is the name of the field +// Descending reverse the sort order (default false) +// Type allows forcing of string/number/date behavior (default auto) +// Mode controls behavior for multi-values fields (default first) +// Missing controls behavior of missing values (default last) +type SortField struct { + Field string + Desc bool + Type SortFieldType + Mode SortFieldMode + Missing SortFieldMissing +} + +// Value returns the sort value of the DocumentMatch +func (s *SortField) Value(i *DocumentMatch) string { + iTerms := i.CachedFieldTerms[s.Field] + iTerms = s.filterTermsByType(iTerms) + iTerm := s.filterTermsByMode(iTerms) + return iTerm +} + +// Descending determines the order of the sort +func (s *SortField) Descending() bool { + return s.Desc +} + +func (s *SortField) filterTermsByMode(terms []string) string { + if len(terms) == 1 || (len(terms) > 1 && s.Mode == SortFieldDefault) { + return terms[0] + } else if len(terms) > 1 { + switch s.Mode { + case SortFieldMin: + sort.Strings(terms) + return terms[0] + case SortFieldMax: + sort.Strings(terms) + return terms[len(terms)-1] + } + } + + // handle missing terms + if s.Missing == SortFieldMissingLast { + if s.Desc { + return LowTerm + } + return HighTerm + } + if s.Desc { + return HighTerm + } + return LowTerm +} + +// filterTermsByType attempts to make one pass on the terms +// if we are in auto-mode AND all the terms look like prefix-coded numbers +// return only the terms which had shift of 0 +// if we are in explicit number or date mode, return only valid +// prefix coded numbers with shift of 0 +func (s *SortField) filterTermsByType(terms []string) []string { + stype := s.Type + if stype == SortFieldAuto { + allTermsPrefixCoded := true + var termsWithShiftZero []string + for _, term := range terms { + valid, shift := numeric_util.ValidPrefixCodedTerm(term) + if valid && shift == 0 { + termsWithShiftZero = append(termsWithShiftZero, term) + } else if !valid { + allTermsPrefixCoded = false + } + } + if allTermsPrefixCoded { + terms = termsWithShiftZero + } + } else if stype == SortFieldAsNumber || stype == SortFieldAsDate { + var termsWithShiftZero []string + for _, term := range terms { + valid, shift := numeric_util.ValidPrefixCodedTerm(term) + if valid && shift == 0 { + termsWithShiftZero = append(termsWithShiftZero) + } + } + terms = termsWithShiftZero + } + return terms +} + +// RequiresDocID says this SearchSort does not require the DocID be loaded +func (s *SortField) RequiresDocID() bool { return false } + +// RequiresScoring says this SearchStore does not require scoring +func (s *SortField) RequiresScoring() bool { return false } + +// RequiresFields says this SearchStore requires the specified stored field +func (s *SortField) RequiresFields() []string { return []string{s.Field} } + +func (s *SortField) MarshalJSON() ([]byte, error) { + // see if simple format can be used + if s.Missing == SortFieldMissingLast && + s.Mode == SortFieldDefault && + s.Type == SortFieldAuto { + if s.Desc { + return json.Marshal("-" + s.Field) + } + return json.Marshal(s.Field) + } + sfm := map[string]interface{}{ + "by": "field", + "field": s.Field, + } + if s.Desc { + sfm["desc"] = true + } + if s.Missing > SortFieldMissingLast { + switch s.Missing { + case SortFieldMissingFirst: + sfm["missing"] = "first" + } + } + if s.Mode > SortFieldDefault { + switch s.Mode { + case SortFieldMin: + sfm["mode"] = "min" + case SortFieldMax: + sfm["mode"] = "max" + } + } + if s.Type > SortFieldAuto { + switch s.Type { + case SortFieldAsString: + sfm["type"] = "string" + case SortFieldAsNumber: + sfm["type"] = "number" + case SortFieldAsDate: + sfm["type"] = "date" + } + } + + return json.Marshal(sfm) +} + +// SortDocID will sort results by the document identifier +type SortDocID struct { + Desc bool +} + +// Value returns the sort value of the DocumentMatch +func (s *SortDocID) Value(i *DocumentMatch) string { + return i.ID +} + +// Descending determines the order of the sort +func (s *SortDocID) Descending() bool { + return s.Desc +} + +// RequiresDocID says this SearchSort does require the DocID be loaded +func (s *SortDocID) RequiresDocID() bool { return true } + +// RequiresScoring says this SearchStore does not require scoring +func (s *SortDocID) RequiresScoring() bool { return false } + +// RequiresFields says this SearchStore does not require any stored fields +func (s *SortDocID) RequiresFields() []string { return nil } + +func (s *SortDocID) MarshalJSON() ([]byte, error) { + if s.Desc { + return json.Marshal("-_id") + } + return json.Marshal("_id") +} + +// SortScore will sort results by the document match score +type SortScore struct { + Desc bool +} + +// Value returns the sort value of the DocumentMatch +func (s *SortScore) Value(i *DocumentMatch) string { + return "_score" +} + +// Descending determines the order of the sort +func (s *SortScore) Descending() bool { + return s.Desc +} + +// RequiresDocID says this SearchSort does not require the DocID be loaded +func (s *SortScore) RequiresDocID() bool { return false } + +// RequiresScoring says this SearchStore does require scoring +func (s *SortScore) RequiresScoring() bool { return true } + +// RequiresFields says this SearchStore does not require any store fields +func (s *SortScore) RequiresFields() []string { return nil } + +func (s *SortScore) MarshalJSON() ([]byte, error) { + if s.Desc { + return json.Marshal("-_score") + } + return json.Marshal("_score") +} diff --git a/vendor/github.com/blevesearch/segment/.travis.yml b/vendor/github.com/blevesearch/segment/.travis.yml index d032f23..b9d58e7 100644 --- a/vendor/github.com/blevesearch/segment/.travis.yml +++ b/vendor/github.com/blevesearch/segment/.travis.yml @@ -1,10 +1,9 @@ language: go go: - - 1.4 + - 1.7 script: - - go get golang.org/x/tools/cmd/vet - go get golang.org/x/tools/cmd/cover - go get github.com/mattn/goveralls - go test -v -covermode=count -coverprofile=profile.out diff --git a/vendor/github.com/boltdb/bolt/README.md b/vendor/github.com/boltdb/bolt/README.md index 3bff3cc..8523e33 100644 --- a/vendor/github.com/boltdb/bolt/README.md +++ b/vendor/github.com/boltdb/bolt/README.md @@ -1,4 +1,4 @@ -Bolt [![Coverage Status](https://coveralls.io/repos/boltdb/bolt/badge.svg?branch=master)](https://coveralls.io/r/boltdb/bolt?branch=master) [![GoDoc](https://godoc.org/github.com/boltdb/bolt?status.svg)](https://godoc.org/github.com/boltdb/bolt) ![Version](https://img.shields.io/badge/version-1.0-green.svg) +Bolt [![Coverage Status](https://coveralls.io/repos/boltdb/bolt/badge.svg?branch=master)](https://coveralls.io/r/boltdb/bolt?branch=master) [![GoDoc](https://godoc.org/github.com/boltdb/bolt?status.svg)](https://godoc.org/github.com/boltdb/bolt) ![Version](https://img.shields.io/badge/version-1.2.1-green.svg) ==== Bolt is a pure Go key/value store inspired by [Howard Chu's][hyc_symas] @@ -313,7 +313,7 @@ func (s *Store) CreateUser(u *User) error { // Generate ID for the user. // This returns an error only if the Tx is closed or not writeable. // That can't happen in an Update() call so I ignore the error check. - id, _ = b.NextSequence() + id, _ := b.NextSequence() u.ID = int(id) // Marshal user data into bytes. @@ -557,7 +557,7 @@ if err != nil { Bolt is able to run on mobile devices by leveraging the binding feature of the [gomobile](https://github.com/golang/mobile) tool. Create a struct that will contain your database logic and a reference to a `*bolt.DB` with a initializing -contstructor that takes in a filepath where the database file will be stored. +constructor that takes in a filepath where the database file will be stored. Neither Android nor iOS require extra permissions or cleanup from using this method. ```go @@ -807,6 +807,7 @@ them via pull request. Below is a list of public, open source projects that use Bolt: +* [BoltDbWeb](https://github.com/evnix/boltdbweb) - A web based GUI for BoltDB files. * [Operation Go: A Routine Mission](http://gocode.io) - An online programming game for Golang using Bolt for user accounts and a leaderboard. * [Bazil](https://bazil.org/) - A file system that lets your data reside where it is most convenient for it to reside. * [DVID](https://github.com/janelia-flyem/dvid) - Added Bolt as optional storage engine and testing it against Basho-tuned leveldb. @@ -825,7 +826,6 @@ Below is a list of public, open source projects that use Bolt: * [cayley](https://github.com/google/cayley) - Cayley is an open-source graph database using Bolt as optional backend. * [bleve](http://www.blevesearch.com/) - A pure Go search engine similar to ElasticSearch that uses Bolt as the default storage backend. * [tentacool](https://github.com/optiflows/tentacool) - REST api server to manage system stuff (IP, DNS, Gateway...) on a linux server. -* [SkyDB](https://github.com/skydb/sky) - Behavioral analytics database. * [Seaweed File System](https://github.com/chrislusf/seaweedfs) - Highly scalable distributed key~file system with O(1) disk read. * [InfluxDB](https://influxdata.com) - Scalable datastore for metrics, events, and real-time analytics. * [Freehold](http://tshannon.bitbucket.org/freehold/) - An open, secure, and lightweight platform for your files and data. @@ -842,9 +842,11 @@ Below is a list of public, open source projects that use Bolt: * [Go Report Card](https://goreportcard.com/) - Go code quality report cards as a (free and open source) service. * [Boltdb Boilerplate](https://github.com/bobintornado/boltdb-boilerplate) - Boilerplate wrapper around bolt aiming to make simple calls one-liners. * [lru](https://github.com/crowdriff/lru) - Easy to use Bolt-backed Least-Recently-Used (LRU) read-through cache with chainable remote stores. -* [Storm](https://github.com/asdine/storm) - A simple ORM around BoltDB. +* [Storm](https://github.com/asdine/storm) - Simple and powerful ORM for BoltDB. * [GoWebApp](https://github.com/josephspurrier/gowebapp) - A basic MVC web application in Go using BoltDB. * [SimpleBolt](https://github.com/xyproto/simplebolt) - A simple way to use BoltDB. Deals mainly with strings. * [Algernon](https://github.com/xyproto/algernon) - A HTTP/2 web server with built-in support for Lua. Uses BoltDB as the default database backend. +* [MuLiFS](https://github.com/dankomiocevic/mulifs) - Music Library Filesystem creates a filesystem to organise your music files. +* [GoShort](https://github.com/pankajkhairnar/goShort) - GoShort is a URL shortener written in Golang and BoltDB for persistent key/value storage and for routing it's using high performent HTTPRouter. If you are using Bolt in a project please send a pull request to add it to the list. diff --git a/vendor/github.com/boltdb/bolt/freelist.go b/vendor/github.com/boltdb/bolt/freelist.go index 0161948..1b7ba91 100644 --- a/vendor/github.com/boltdb/bolt/freelist.go +++ b/vendor/github.com/boltdb/bolt/freelist.go @@ -166,12 +166,16 @@ func (f *freelist) read(p *page) { } // Copy the list of page ids from the freelist. - ids := ((*[maxAllocSize]pgid)(unsafe.Pointer(&p.ptr)))[idx:count] - f.ids = make([]pgid, len(ids)) - copy(f.ids, ids) + if count == 0 { + f.ids = nil + } else { + ids := ((*[maxAllocSize]pgid)(unsafe.Pointer(&p.ptr)))[idx:count] + f.ids = make([]pgid, len(ids)) + copy(f.ids, ids) - // Make sure they're sorted. - sort.Sort(pgids(f.ids)) + // Make sure they're sorted. + sort.Sort(pgids(f.ids)) + } // Rebuild the page cache. f.reindex() @@ -189,7 +193,9 @@ func (f *freelist) write(p *page) error { // The page.count can only hold up to 64k elements so if we overflow that // number then we handle it by putting the size in the first element. - if len(ids) < 0xFFFF { + if len(ids) == 0 { + p.count = uint16(len(ids)) + } else if len(ids) < 0xFFFF { p.count = uint16(len(ids)) copy(((*[maxAllocSize]pgid)(unsafe.Pointer(&p.ptr)))[:], ids) } else { diff --git a/vendor/github.com/boltdb/bolt/node.go b/vendor/github.com/boltdb/bolt/node.go index e9d64af..159318b 100644 --- a/vendor/github.com/boltdb/bolt/node.go +++ b/vendor/github.com/boltdb/bolt/node.go @@ -201,6 +201,11 @@ func (n *node) write(p *page) { } p.count = uint16(len(n.inodes)) + // Stop here if there are no items to write. + if p.count == 0 { + return + } + // Loop over each item and write it to the page. b := (*[maxAllocSize]byte)(unsafe.Pointer(&p.ptr))[n.pageElementSize()*len(n.inodes):] for i, item := range n.inodes { diff --git a/vendor/github.com/boltdb/bolt/page.go b/vendor/github.com/boltdb/bolt/page.go index 4a55528..7651a6b 100644 --- a/vendor/github.com/boltdb/bolt/page.go +++ b/vendor/github.com/boltdb/bolt/page.go @@ -62,6 +62,9 @@ func (p *page) leafPageElement(index uint16) *leafPageElement { // leafPageElements retrieves a list of leaf nodes. func (p *page) leafPageElements() []leafPageElement { + if p.count == 0 { + return nil + } return ((*[0x7FFFFFF]leafPageElement)(unsafe.Pointer(&p.ptr)))[:] } @@ -72,6 +75,9 @@ func (p *page) branchPageElement(index uint16) *branchPageElement { // branchPageElements retrieves a list of branch nodes. func (p *page) branchPageElements() []branchPageElement { + if p.count == 0 { + return nil + } return ((*[0x7FFFFFF]branchPageElement)(unsafe.Pointer(&p.ptr)))[:] } diff --git a/vendor/github.com/willf/bitset/.gitignore b/vendor/github.com/willf/bitset/.gitignore deleted file mode 100644 index 5c204d2..0000000 --- a/vendor/github.com/willf/bitset/.gitignore +++ /dev/null @@ -1,26 +0,0 @@ -# Compiled Object files, Static and Dynamic libs (Shared Objects) -*.o -*.a -*.so - -# Folders -_obj -_test - -# Architecture specific extensions/prefixes -*.[568vq] -[568vq].out - -*.cgo1.go -*.cgo2.c -_cgo_defun.c -_cgo_gotypes.go -_cgo_export.* - -_testmain.go - -*.exe -*.test -*.prof - -target diff --git a/vendor/github.com/willf/bitset/.travis.yml b/vendor/github.com/willf/bitset/.travis.yml deleted file mode 100644 index 11a99a2..0000000 --- a/vendor/github.com/willf/bitset/.travis.yml +++ /dev/null @@ -1,16 +0,0 @@ -language: go - -sudo: false - -branches: - except: - - release - -branches: - only: - - master - - develop - -go: - - 1.5 - - tip diff --git a/vendor/github.com/willf/bitset/LICENSE b/vendor/github.com/willf/bitset/LICENSE deleted file mode 100644 index 59cab8a..0000000 --- a/vendor/github.com/willf/bitset/LICENSE +++ /dev/null @@ -1,27 +0,0 @@ -Copyright (c) 2014 Will Fitzgerald. All rights reserved. - -Redistribution and use in source and binary forms, with or without -modification, are permitted provided that the following conditions are -met: - - * Redistributions of source code must retain the above copyright -notice, this list of conditions and the following disclaimer. - * Redistributions in binary form must reproduce the above -copyright notice, this list of conditions and the following disclaimer -in the documentation and/or other materials provided with the -distribution. - * Neither the name of Google Inc. nor the names of its -contributors may be used to endorse or promote products derived from -this software without specific prior written permission. - -THIS SOFTWARE IS PROVIDED BY THE COPYRIGHT HOLDERS AND CONTRIBUTORS -"AS IS" AND ANY EXPRESS OR IMPLIED WARRANTIES, INCLUDING, BUT NOT -LIMITED TO, THE IMPLIED WARRANTIES OF MERCHANTABILITY AND FITNESS FOR -A PARTICULAR PURPOSE ARE DISCLAIMED. IN NO EVENT SHALL THE COPYRIGHT -OWNER OR CONTRIBUTORS BE LIABLE FOR ANY DIRECT, INDIRECT, INCIDENTAL, -SPECIAL, EXEMPLARY, OR CONSEQUENTIAL DAMAGES (INCLUDING, BUT NOT -LIMITED TO, PROCUREMENT OF SUBSTITUTE GOODS OR SERVICES; LOSS OF USE, -DATA, OR PROFITS; OR BUSINESS INTERRUPTION) HOWEVER CAUSED AND ON ANY -THEORY OF LIABILITY, WHETHER IN CONTRACT, STRICT LIABILITY, OR TORT -(INCLUDING NEGLIGENCE OR OTHERWISE) ARISING IN ANY WAY OUT OF THE USE -OF THIS SOFTWARE, EVEN IF ADVISED OF THE POSSIBILITY OF SUCH DAMAGE. diff --git a/vendor/github.com/willf/bitset/Makefile b/vendor/github.com/willf/bitset/Makefile deleted file mode 100644 index a4d3f2b..0000000 --- a/vendor/github.com/willf/bitset/Makefile +++ /dev/null @@ -1,109 +0,0 @@ -# MAKEFILE -# -# @author Nicola Asuni -# @link https://github.com/willf/bitset -# ------------------------------------------------------------------------------ - -# List special make targets that are not associated with files -.PHONY: help all qa test format fmtcheck vet lint coverage docs deps clean nuke - -# Use bash as shell (Note: Ubuntu now uses dash which doesn't support PIPESTATUS). -SHELL=/bin/bash - -# Project owner -OWNER=willf - -# Project name -PROJECT=bitset - -# Name of RPM or DEB package -PKGNAME=${OWNER}-${PROJECT} - -# Go lang path. Set if necessary -# GOPATH=$(shell readlink -f $(shell pwd)/../../../../) - -# Current directory -CURRENTDIR=$(shell pwd) - -# --- MAKE TARGETS --- - -# Display general help about this command -help: - @echo "" - @echo "$(PROJECT) Makefile." - @echo "The following commands are available:" - @echo "" - @echo " make qa : Run all the tests" - @echo " make test : Run the unit tests" - @echo "" - @echo " make format : Format the source code" - @echo " make fmtcheck : Check if the source code has been formatted" - @echo " make vet : Check for syntax errors" - @echo " make lint : Check for style errors" - @echo " make coverage : Generate the coverage report" - @echo "" - @echo " make docs : Generate source code documentation" - @echo "" - @echo " make deps : Get the dependencies" - @echo " make clean : Remove any build artifact" - @echo " make nuke : Deletes any intermediate file" - @echo "" - -# Alias for help target -all: help - -# Run the unit tests -test: - @mkdir -p target/test - @mkdir -p target/report - GOPATH=$(GOPATH) go test -covermode=count -coverprofile=target/report/coverage.out -bench=. -race -v ./... | tee >(PATH=$(GOPATH)/bin:$(PATH) go-junit-report > target/test/report.xml); test $${PIPESTATUS[0]} -eq 0 - -# Format the source code -format: - @find ./ -type f -name "*.go" -exec gofmt -w {} \; - -# Check if the source code has been formatted -fmtcheck: - @mkdir -p target - @find ./ -type f -name "*.go" -exec gofmt -d {} \; | tee target/format.diff - @test ! -s target/format.diff || { echo "ERROR: the source code has not been formatted - please use 'make format' or 'gofmt'"; exit 1; } - -# Check for syntax errors -vet: - GOPATH=$(GOPATH) go vet ./... - -# Check for style errors -lint: - GOPATH=$(GOPATH) PATH=$(GOPATH)/bin:$(PATH) golint ./... - -# Generate the coverage report -coverage: - GOPATH=$(GOPATH) go tool cover -html=target/report/coverage.out -o target/report/coverage.html - -# Generate source docs -docs: - @mkdir -p target/docs - nohup sh -c 'GOPATH=$(GOPATH) godoc -http=127.0.0.1:6060' > target/godoc_server.log 2>&1 & - wget --directory-prefix=target/docs/ --execute robots=off --retry-connrefused --recursive --no-parent --adjust-extension --page-requisites --convert-links http://127.0.0.1:6060/pkg/github.com/${OWNER}/${PROJECT}/ ; kill -9 `lsof -ti :6060` - @echo ''${PKGNAME}' Documentation ...' > target/docs/index.html - -# Alias to run targets: fmtcheck test vet lint coverage -qa: fmtcheck test vet lint coverage - -# --- INSTALL --- - -# Get the dependencies -deps: - GOPATH=$(GOPATH) go get ./... - GOPATH=$(GOPATH) go get github.com/golang/lint/golint - GOPATH=$(GOPATH) go get github.com/jstemmer/go-junit-report - GOPATH=$(GOPATH) go get github.com/axw/gocov/gocov - -# Remove any build artifact -clean: - GOPATH=$(GOPATH) go clean ./... - -# Deletes any intermediate file -nuke: - rm -rf ./target - GOPATH=$(GOPATH) go clean -i ./... diff --git a/vendor/github.com/willf/bitset/README.md b/vendor/github.com/willf/bitset/README.md deleted file mode 100644 index 2fe69a1..0000000 --- a/vendor/github.com/willf/bitset/README.md +++ /dev/null @@ -1,65 +0,0 @@ -# bitset - -*Go language library to map between non-negative integers and boolean values* - -[![Master Branch](https://img.shields.io/badge/-master:-gray.svg)](https://github.com/willf/bitset/tree/master) -[![Master Build Status](https://travis-ci.org/willf/bitset.svg?branch=master)](https://travis-ci.org/willf/bitset) -[![Develop Branch](https://img.shields.io/badge/-develop:-gray.svg)](https://github.com/willf/bitset/tree/develop) -[![Develop Build Status](https://secure.travis-ci.org/willf/bitset.svg?branch=develop)](https://travis-ci.org/willf/bitset?branch=develop) - - - -## Description - -Package bitset implements bitsets, a mapping between non-negative integers and boolean values. -It should be more efficient than map[uint] bool. - -It provides methods for setting, clearing, flipping, and testing individual integers. - -But it also provides set intersection, union, difference, complement, and symmetric operations, as well as tests to check whether any, all, or no bits are set, and querying a bitset's current length and number of postive bits. - -BitSets are expanded to the size of the largest set bit; the memory allocation is approximately Max bits, where Max is the largest set bit. BitSets are never shrunk. On creation, a hint can be given for the number of bits that will be used. - -Many of the methods, including Set, Clear, and Flip, return a BitSet pointer, which allows for chaining. - -### Example use: - - import "bitset" - var b BitSet - b.Set(10).Set(11) - if b.Test(1000) { - b.Clear(1000) - } - for i,e := v.NextSet(0); e; i,e = v.NextSet(i + 1) { - frmt.Println("The following bit is set:",i); - } - if B.Intersection(bitset.New(100).Set(10)).Count() > 1 { - fmt.Println("Intersection works.") - } - -As an alternative to BitSets, one should check out the 'big' package, which provides a (less set-theoretical) view of bitsets. - -Discussions golang-nuts Google Group: - -* [Revised BitSet](https://groups.google.com/forum/#!topic/golang-nuts/5i3l0CXDiBg) -* [simple bitset?](https://groups.google.com/d/topic/golang-nuts/7n1VkRTlBf4/discussion) - -Godoc documentation is at: https://godoc.org/github.com/willf/bitset - - -## Getting started - -This application is written in the go language, please refer to the guides in https://golang.org for getting started. - -This project include a Makefile that allows you to test and build the project with simple commands. -To see all available options: -```bash -make help -``` - -## Running all tests - -Before committing the code, please check if it passes all tests using -```bash -make qa -``` diff --git a/vendor/github.com/willf/bitset/RELEASE b/vendor/github.com/willf/bitset/RELEASE deleted file mode 100644 index 56a6051..0000000 --- a/vendor/github.com/willf/bitset/RELEASE +++ /dev/null @@ -1 +0,0 @@ -1 \ No newline at end of file diff --git a/vendor/github.com/willf/bitset/VERSION b/vendor/github.com/willf/bitset/VERSION deleted file mode 100644 index 9084fa2..0000000 --- a/vendor/github.com/willf/bitset/VERSION +++ /dev/null @@ -1 +0,0 @@ -1.1.0 diff --git a/vendor/github.com/willf/bitset/bitset.go b/vendor/github.com/willf/bitset/bitset.go deleted file mode 100644 index 1bc27f2..0000000 --- a/vendor/github.com/willf/bitset/bitset.go +++ /dev/null @@ -1,650 +0,0 @@ -/* -Package bitset implements bitsets, a mapping -between non-negative integers and boolean values. It should be more -efficient than map[uint] bool. - -It provides methods for setting, clearing, flipping, and testing -individual integers. - -But it also provides set intersection, union, difference, -complement, and symmetric operations, as well as tests to -check whether any, all, or no bits are set, and querying a -bitset's current length and number of postive bits. - -BitSets are expanded to the size of the largest set bit; the -memory allocation is approximately Max bits, where Max is -the largest set bit. BitSets are never shrunk. On creation, -a hint can be given for the number of bits that will be used. - -Many of the methods, including Set,Clear, and Flip, return -a BitSet pointer, which allows for chaining. - -Example use: - - import "bitset" - var b BitSet - b.Set(10).Set(11) - if b.Test(1000) { - b.Clear(1000) - } - if B.Intersection(bitset.New(100).Set(10)).Count() > 1 { - fmt.Println("Intersection works.") - } - -As an alternative to BitSets, one should check out the 'big' package, -which provides a (less set-theoretical) view of bitsets. - -*/ -package bitset - -import ( - "bufio" - "bytes" - "encoding/base64" - "encoding/binary" - "encoding/json" - "errors" - "fmt" - "io" -) - -// the wordSize of a bit set -const wordSize = uint(64) - -// log2WordSize is lg(wordSize) -const log2WordSize = uint(6) - -// A BitSet is a set of bits. The zero value of a BitSet is an empty set of length 0. -type BitSet struct { - length uint - set []uint64 -} - -// Error is used to distinguish errors (panics) generated in this package. -type Error string - -// safeSet will fixup b.set to be non-nil and return the field value -func (b *BitSet) safeSet() []uint64 { - if b.set == nil { - b.set = make([]uint64, wordsNeeded(0)) - } - return b.set -} - -// From is a constructor used to create a BitSet from an array of integers -func From(buf []uint64) *BitSet { - return &BitSet{uint(len(buf)) * 64, buf} -} - -// Bytes returns the bitset as array of integers -func (b *BitSet) Bytes() []uint64 { - return b.set -} - -// wordsNeeded calculates the number of words needed for i bits -func wordsNeeded(i uint) int { - if i > ((^uint(0)) - wordSize + 1) { - return int((^uint(0)) >> log2WordSize) - } - return int((i + (wordSize - 1)) >> log2WordSize) -} - -// New creates a new BitSet with a hint that length bits will be required -func New(length uint) *BitSet { - return &BitSet{length, make([]uint64, wordsNeeded(length))} -} - -// Cap returns the total possible capicity, or number of bits -func Cap() uint { - return ^uint(0) -} - -// Len returns the length of the BitSet in words -func (b *BitSet) Len() uint { - return b.length -} - -// extendSetMaybe adds additional words to incorporate new bits if needed -func (b *BitSet) extendSetMaybe(i uint) { - if i >= b.length { // if we need more bits, make 'em - nsize := wordsNeeded(i + 1) - if b.set == nil { - b.set = make([]uint64, nsize) - } else if len(b.set) < nsize { - newset := make([]uint64, nsize) - copy(newset, b.set) - b.set = newset - } - b.length = i + 1 - } -} - -// Test whether bit i is set. -func (b *BitSet) Test(i uint) bool { - if i >= b.length { - return false - } - return b.set[i>>log2WordSize]&(1<<(i&(wordSize-1))) != 0 -} - -// Set bit i to 1 -func (b *BitSet) Set(i uint) *BitSet { - b.extendSetMaybe(i) - b.set[i>>log2WordSize] |= 1 << (i & (wordSize - 1)) - return b -} - -// Clear bit i to 0 -func (b *BitSet) Clear(i uint) *BitSet { - if i >= b.length { - return b - } - b.set[i>>log2WordSize] &^= 1 << (i & (wordSize - 1)) - return b -} - -// SetTo sets bit i to value -func (b *BitSet) SetTo(i uint, value bool) *BitSet { - if value { - return b.Set(i) - } - return b.Clear(i) -} - -// Flip bit at i -func (b *BitSet) Flip(i uint) *BitSet { - if i >= b.length { - return b.Set(i) - } - b.set[i>>log2WordSize] ^= 1 << (i & (wordSize - 1)) - return b -} - -// NextSet returns the next bit set from the specified index, -// including possibly the current index -// along with an error code (true = valid, false = no set bit found) -// for i,e := v.NextSet(0); e; i,e = v.NextSet(i + 1) {...} -func (b *BitSet) NextSet(i uint) (uint, bool) { - x := int(i >> log2WordSize) - if x >= len(b.set) { - return 0, false - } - w := b.set[x] - w = w >> (i & (wordSize - 1)) - if w != 0 { - return i + trailingZeroes64(w), true - } - x = x + 1 - for x < len(b.set) { - if b.set[x] != 0 { - return uint(x)*wordSize + trailingZeroes64(b.set[x]), true - } - x = x + 1 - - } - return 0, false -} - -// ClearAll clears the entire BitSet -func (b *BitSet) ClearAll() *BitSet { - if b != nil && b.set != nil { - for i := range b.set { - b.set[i] = 0 - } - } - return b -} - -// wordCount returns the number of words used in a bit set -func (b *BitSet) wordCount() int { - return wordsNeeded(b.length) -} - -// Clone this BitSet -func (b *BitSet) Clone() *BitSet { - c := New(b.length) - if b.set != nil { // Clone should not modify current object - copy(c.set, b.set) - } - return c -} - -// Copy into a destination BitSet -// Returning the size of the destination BitSet -// like array copy -func (b *BitSet) Copy(c *BitSet) (count uint) { - if c == nil { - return - } - if b.set != nil { // Copy should not modify current object - copy(c.set, b.set) - } - count = c.length - if b.length < c.length { - count = b.length - } - return -} - -// Count (number of set bits) -func (b *BitSet) Count() uint { - if b != nil && b.set != nil { - return uint(popcntSlice(b.set)) - } - return 0 -} - -var deBruijn = [...]byte{ - 0, 1, 56, 2, 57, 49, 28, 3, 61, 58, 42, 50, 38, 29, 17, 4, - 62, 47, 59, 36, 45, 43, 51, 22, 53, 39, 33, 30, 24, 18, 12, 5, - 63, 55, 48, 27, 60, 41, 37, 16, 46, 35, 44, 21, 52, 32, 23, 11, - 54, 26, 40, 15, 34, 20, 31, 10, 25, 14, 19, 9, 13, 8, 7, 6, -} - -func trailingZeroes64(v uint64) uint { - return uint(deBruijn[((v&-v)*0x03f79d71b4ca8b09)>>58]) -} - -// Equal tests the equvalence of two BitSets. -// False if they are of different sizes, otherwise true -// only if all the same bits are set -func (b *BitSet) Equal(c *BitSet) bool { - if c == nil { - return false - } - if b.length != c.length { - return false - } - if b.length == 0 { // if they have both length == 0, then could have nil set - return true - } - // testing for equality shoud not transform the bitset (no call to safeSet) - - for p, v := range b.set { - if c.set[p] != v { - return false - } - } - return true -} - -func panicIfNull(b *BitSet) { - if b == nil { - panic(Error("BitSet must not be null")) - } -} - -// Difference of base set and other set -// This is the BitSet equivalent of &^ (and not) -func (b *BitSet) Difference(compare *BitSet) (result *BitSet) { - panicIfNull(b) - panicIfNull(compare) - result = b.Clone() // clone b (in case b is bigger than compare) - l := int(compare.wordCount()) - if l > int(b.wordCount()) { - l = int(b.wordCount()) - } - for i := 0; i < l; i++ { - result.set[i] = b.set[i] &^ compare.set[i] - } - return -} - -// DifferenceCardinality computes the cardinality of the differnce -func (b *BitSet) DifferenceCardinality(compare *BitSet) uint { - panicIfNull(b) - panicIfNull(compare) - l := int(compare.wordCount()) - if l > int(b.wordCount()) { - l = int(b.wordCount()) - } - cnt := uint64(0) - cnt += popcntMaskSlice(b.set[:l], compare.set[:l]) - cnt += popcntSlice(b.set[l:]) - return uint(cnt) -} - -// InPlaceDifference computes the difference of base set and other set -// This is the BitSet equivalent of &^ (and not) -func (b *BitSet) InPlaceDifference(compare *BitSet) { - panicIfNull(b) - panicIfNull(compare) - l := int(compare.wordCount()) - if l > int(b.wordCount()) { - l = int(b.wordCount()) - } - for i := 0; i < l; i++ { - b.set[i] &^= compare.set[i] - } -} - -// Convenience function: return two bitsets ordered by -// increasing length. Note: neither can be nil -func sortByLength(a *BitSet, b *BitSet) (ap *BitSet, bp *BitSet) { - if a.length <= b.length { - ap, bp = a, b - } else { - ap, bp = b, a - } - return -} - -// Intersection of base set and other set -// This is the BitSet equivalent of & (and) -func (b *BitSet) Intersection(compare *BitSet) (result *BitSet) { - panicIfNull(b) - panicIfNull(compare) - b, compare = sortByLength(b, compare) - result = New(b.length) - for i, word := range b.set { - result.set[i] = word & compare.set[i] - } - return -} - -// IntersectionCardinality computes the cardinality of the union -func (b *BitSet) IntersectionCardinality(compare *BitSet) uint { - panicIfNull(b) - panicIfNull(compare) - b, compare = sortByLength(b, compare) - cnt := popcntAndSlice(b.set, compare.set) - return uint(cnt) -} - -// InPlaceIntersection destructively computes the intersection of -// base set and the compare set. -// This is the BitSet equivalent of & (and) -func (b *BitSet) InPlaceIntersection(compare *BitSet) { - panicIfNull(b) - panicIfNull(compare) - l := int(compare.wordCount()) - if l > int(b.wordCount()) { - l = int(b.wordCount()) - } - for i := 0; i < l; i++ { - b.set[i] &= compare.set[i] - } - for i := l; i < len(b.set); i++ { - b.set[i] = 0 - } - if compare.length > 0 { - b.extendSetMaybe(compare.length - 1) - } - return -} - -// Union of base set and other set -// This is the BitSet equivalent of | (or) -func (b *BitSet) Union(compare *BitSet) (result *BitSet) { - panicIfNull(b) - panicIfNull(compare) - b, compare = sortByLength(b, compare) - result = compare.Clone() - for i, word := range b.set { - result.set[i] = word | compare.set[i] - } - return -} - -// UnionCardinality computes the cardinality of the uniton of the base set -// and the compare set. -func (b *BitSet) UnionCardinality(compare *BitSet) uint { - panicIfNull(b) - panicIfNull(compare) - b, compare = sortByLength(b, compare) - cnt := popcntOrSlice(b.set, compare.set) - if len(compare.set) > len(b.set) { - cnt += popcntSlice(compare.set[len(b.set):]) - } - return uint(cnt) -} - -// InPlaceUnion creates the destructive union of base set and compare set. -// This is the BitSet equivalent of | (or). -func (b *BitSet) InPlaceUnion(compare *BitSet) { - panicIfNull(b) - panicIfNull(compare) - l := int(compare.wordCount()) - if l > int(b.wordCount()) { - l = int(b.wordCount()) - } - if compare.length > 0 { - b.extendSetMaybe(compare.length - 1) - } - for i := 0; i < l; i++ { - b.set[i] |= compare.set[i] - } - if len(compare.set) > l { - for i := l; i < len(compare.set); i++ { - b.set[i] = compare.set[i] - } - } -} - -// SymmetricDifference of base set and other set -// This is the BitSet equivalent of ^ (xor) -func (b *BitSet) SymmetricDifference(compare *BitSet) (result *BitSet) { - panicIfNull(b) - panicIfNull(compare) - b, compare = sortByLength(b, compare) - // compare is bigger, so clone it - result = compare.Clone() - for i, word := range b.set { - result.set[i] = word ^ compare.set[i] - } - return -} - -// SymmetricDifferenceCardinality computes the cardinality of the symmetric difference -func (b *BitSet) SymmetricDifferenceCardinality(compare *BitSet) uint { - panicIfNull(b) - panicIfNull(compare) - b, compare = sortByLength(b, compare) - cnt := popcntXorSlice(b.set, compare.set) - if len(compare.set) > len(b.set) { - cnt += popcntSlice(compare.set[len(b.set):]) - } - return uint(cnt) -} - -// InPlaceSymmetricDifference creates the destructive SymmetricDifference of base set and other set -// This is the BitSet equivalent of ^ (xor) -func (b *BitSet) InPlaceSymmetricDifference(compare *BitSet) { - panicIfNull(b) - panicIfNull(compare) - l := int(compare.wordCount()) - if l > int(b.wordCount()) { - l = int(b.wordCount()) - } - if compare.length > 0 { - b.extendSetMaybe(compare.length - 1) - } - for i := 0; i < l; i++ { - b.set[i] ^= compare.set[i] - } - if len(compare.set) > l { - for i := l; i < len(compare.set); i++ { - b.set[i] = compare.set[i] - } - } -} - -// Is the length an exact multiple of word sizes? -func (b *BitSet) isEven() bool { - return b.length%wordSize == 0 -} - -// Clean last word by setting unused bits to 0 -func (b *BitSet) cleanLastWord() { - if !b.isEven() { - // Mask for cleaning last word - const allBits uint64 = 0xffffffffffffffff - b.set[wordsNeeded(b.length)-1] &= allBits >> (wordSize - b.length%wordSize) - } -} - -// Complement computes the (local) complement of a biset (up to length bits) -func (b *BitSet) Complement() (result *BitSet) { - panicIfNull(b) - result = New(b.length) - for i, word := range b.set { - result.set[i] = ^word - } - result.cleanLastWord() - return -} - -// All returns true if all bits are set, false otherwise -func (b *BitSet) All() bool { - panicIfNull(b) - return b.Count() == b.length -} - -// None returns true if no bit is set, false otherwise -func (b *BitSet) None() bool { - panicIfNull(b) - if b != nil && b.set != nil { - for _, word := range b.set { - if word > 0 { - return false - } - } - return true - } - return true -} - -// Any returns true if any bit is set, false otherwise -func (b *BitSet) Any() bool { - panicIfNull(b) - return !b.None() -} - -// IsSuperSet returns true if this is a superset of the other set -func (b *BitSet) IsSuperSet(other *BitSet) bool { - for i, e := other.NextSet(0); e; i, e = other.NextSet(i + 1) { - if !b.Test(i) { - return false - } - } - return true -} - -// IsStrictSuperSet returns true if this is a strict superset of the other set -func (b *BitSet) IsStrictSuperSet(other *BitSet) bool { - return b.Count() > other.Count() && b.IsSuperSet(other) -} - -// DumpAsBits dumps a bit set as a string of bits -func (b *BitSet) DumpAsBits() string { - if b.set == nil { - return "." - } - buffer := bytes.NewBufferString("") - i := len(b.set) - 1 - for ; i >= 0; i-- { - fmt.Fprintf(buffer, "%064b.", b.set[i]) - } - return string(buffer.Bytes()) -} - -// BinaryStorageSize returns the binary storage requirements -func (b *BitSet) BinaryStorageSize() int { - return binary.Size(uint64(0)) + binary.Size(b.set) -} - -// WriteTo writes a BitSet to a stream -func (b *BitSet) WriteTo(stream io.Writer) (int64, error) { - length := uint64(b.length) - - // Write length - err := binary.Write(stream, binary.BigEndian, length) - if err != nil { - return 0, err - } - - // Write set - err = binary.Write(stream, binary.BigEndian, b.set) - return int64(b.BinaryStorageSize()), err -} - -// ReadFrom reads a BitSet from a stream written using WriteTo -func (b *BitSet) ReadFrom(stream io.Reader) (int64, error) { - var length uint64 - - // Read length first - err := binary.Read(stream, binary.BigEndian, &length) - if err != nil { - return 0, err - } - newset := New(uint(length)) - - if uint64(newset.length) != length { - return 0, errors.New("Unmarshalling error: type mismatch") - } - - // Read remaining bytes as set - err = binary.Read(stream, binary.BigEndian, newset.set) - if err != nil { - return 0, err - } - - *b = *newset - return int64(b.BinaryStorageSize()), nil -} - -// MarshalBinary encodes a BitSet into a binary form and returns the result. -func (b *BitSet) MarshalBinary() ([]byte, error) { - var buf bytes.Buffer - writer := bufio.NewWriter(&buf) - - _, err := b.WriteTo(writer) - if err != nil { - return []byte{}, err - } - - err = writer.Flush() - - return buf.Bytes(), err -} - -// UnmarshalBinary decodes the binary form generated by MarshalBinary. -func (b *BitSet) UnmarshalBinary(data []byte) error { - buf := bytes.NewReader(data) - reader := bufio.NewReader(buf) - - _, err := b.ReadFrom(reader) - - return err -} - -// MarshalJSON marshals a BitSet as a JSON structure -func (b *BitSet) MarshalJSON() ([]byte, error) { - buffer := bytes.NewBuffer(make([]byte, 0, b.BinaryStorageSize())) - _, err := b.WriteTo(buffer) - if err != nil { - return nil, err - } - - // URLEncode all bytes - return json.Marshal(base64.URLEncoding.EncodeToString(buffer.Bytes())) -} - -// UnmarshalJSON unmarshals a BitSet from JSON created using MarshalJSON -func (b *BitSet) UnmarshalJSON(data []byte) error { - // Unmarshal as string - var s string - err := json.Unmarshal(data, &s) - if err != nil { - return err - } - - // URLDecode string - buf, err := base64.URLEncoding.DecodeString(s) - if err != nil { - return err - } - - _, err = b.ReadFrom(bytes.NewReader(buf)) - return err -} diff --git a/vendor/github.com/willf/bitset/popcnt.go b/vendor/github.com/willf/bitset/popcnt.go deleted file mode 100644 index 76577a8..0000000 --- a/vendor/github.com/willf/bitset/popcnt.go +++ /dev/null @@ -1,53 +0,0 @@ -package bitset - -// bit population count, take from -// https://code.google.com/p/go/issues/detail?id=4988#c11 -// credit: https://code.google.com/u/arnehormann/ -func popcount(x uint64) (n uint64) { - x -= (x >> 1) & 0x5555555555555555 - x = (x>>2)&0x3333333333333333 + x&0x3333333333333333 - x += x >> 4 - x &= 0x0f0f0f0f0f0f0f0f - x *= 0x0101010101010101 - return x >> 56 -} - -func popcntSliceGo(s []uint64) uint64 { - cnt := uint64(0) - for _, x := range s { - cnt += popcount(x) - } - return cnt -} - -func popcntMaskSliceGo(s, m []uint64) uint64 { - cnt := uint64(0) - for i := range s { - cnt += popcount(s[i] &^ m[i]) - } - return cnt -} - -func popcntAndSliceGo(s, m []uint64) uint64 { - cnt := uint64(0) - for i := range s { - cnt += popcount(s[i] & m[i]) - } - return cnt -} - -func popcntOrSliceGo(s, m []uint64) uint64 { - cnt := uint64(0) - for i := range s { - cnt += popcount(s[i] | m[i]) - } - return cnt -} - -func popcntXorSliceGo(s, m []uint64) uint64 { - cnt := uint64(0) - for i := range s { - cnt += popcount(s[i] ^ m[i]) - } - return cnt -} diff --git a/vendor/github.com/willf/bitset/popcnt_amd64.go b/vendor/github.com/willf/bitset/popcnt_amd64.go deleted file mode 100644 index 665a864..0000000 --- a/vendor/github.com/willf/bitset/popcnt_amd64.go +++ /dev/null @@ -1,67 +0,0 @@ -// +build amd64,!appengine - -package bitset - -// *** the following functions are defined in popcnt_amd64.s - -//go:noescape - -func hasAsm() bool - -// useAsm is a flag used to select the GO or ASM implementation of the popcnt function -var useAsm = hasAsm() - -//go:noescape - -func popcntSliceAsm(s []uint64) uint64 - -//go:noescape - -func popcntMaskSliceAsm(s, m []uint64) uint64 - -//go:noescape - -func popcntAndSliceAsm(s, m []uint64) uint64 - -//go:noescape - -func popcntOrSliceAsm(s, m []uint64) uint64 - -//go:noescape - -func popcntXorSliceAsm(s, m []uint64) uint64 - -func popcntSlice(s []uint64) uint64 { - if useAsm { - return popcntSliceAsm(s) - } - return popcntSliceGo(s) -} - -func popcntMaskSlice(s, m []uint64) uint64 { - if useAsm { - return popcntMaskSliceAsm(s, m) - } - return popcntMaskSliceGo(s, m) -} - -func popcntAndSlice(s, m []uint64) uint64 { - if useAsm { - return popcntAndSliceAsm(s, m) - } - return popcntAndSliceGo(s, m) -} - -func popcntOrSlice(s, m []uint64) uint64 { - if useAsm { - return popcntOrSliceAsm(s, m) - } - return popcntOrSliceGo(s, m) -} - -func popcntXorSlice(s, m []uint64) uint64 { - if useAsm { - return popcntXorSliceAsm(s, m) - } - return popcntXorSliceGo(s, m) -} diff --git a/vendor/github.com/willf/bitset/popcnt_amd64.s b/vendor/github.com/willf/bitset/popcnt_amd64.s deleted file mode 100644 index 18f5878..0000000 --- a/vendor/github.com/willf/bitset/popcnt_amd64.s +++ /dev/null @@ -1,103 +0,0 @@ -// +build amd64,!appengine - -TEXT ·hasAsm(SB),4,$0-1 -MOVQ $1, AX -CPUID -SHRQ $23, CX -ANDQ $1, CX -MOVB CX, ret+0(FP) -RET - -#define POPCNTQ_DX_DX BYTE $0xf3; BYTE $0x48; BYTE $0x0f; BYTE $0xb8; BYTE $0xd2 - -TEXT ·popcntSliceAsm(SB),4,$0-32 -XORQ AX, AX -MOVQ s+0(FP), SI -MOVQ s_len+8(FP), CX -TESTQ CX, CX -JZ popcntSliceEnd -popcntSliceLoop: -BYTE $0xf3; BYTE $0x48; BYTE $0x0f; BYTE $0xb8; BYTE $0x16 // POPCNTQ (SI), DX -ADDQ DX, AX -ADDQ $8, SI -LOOP popcntSliceLoop -popcntSliceEnd: -MOVQ AX, ret+24(FP) -RET - -TEXT ·popcntMaskSliceAsm(SB),4,$0-56 -XORQ AX, AX -MOVQ s+0(FP), SI -MOVQ s_len+8(FP), CX -TESTQ CX, CX -JZ popcntMaskSliceEnd -MOVQ m+24(FP), DI -popcntMaskSliceLoop: -MOVQ (DI), DX -NOTQ DX -ANDQ (SI), DX -POPCNTQ_DX_DX -ADDQ DX, AX -ADDQ $8, SI -ADDQ $8, DI -LOOP popcntMaskSliceLoop -popcntMaskSliceEnd: -MOVQ AX, ret+48(FP) -RET - -TEXT ·popcntAndSliceAsm(SB),4,$0-56 -XORQ AX, AX -MOVQ s+0(FP), SI -MOVQ s_len+8(FP), CX -TESTQ CX, CX -JZ popcntAndSliceEnd -MOVQ m+24(FP), DI -popcntAndSliceLoop: -MOVQ (DI), DX -ANDQ (SI), DX -POPCNTQ_DX_DX -ADDQ DX, AX -ADDQ $8, SI -ADDQ $8, DI -LOOP popcntAndSliceLoop -popcntAndSliceEnd: -MOVQ AX, ret+48(FP) -RET - -TEXT ·popcntOrSliceAsm(SB),4,$0-56 -XORQ AX, AX -MOVQ s+0(FP), SI -MOVQ s_len+8(FP), CX -TESTQ CX, CX -JZ popcntOrSliceEnd -MOVQ m+24(FP), DI -popcntOrSliceLoop: -MOVQ (DI), DX -ORQ (SI), DX -POPCNTQ_DX_DX -ADDQ DX, AX -ADDQ $8, SI -ADDQ $8, DI -LOOP popcntOrSliceLoop -popcntOrSliceEnd: -MOVQ AX, ret+48(FP) -RET - -TEXT ·popcntXorSliceAsm(SB),4,$0-56 -XORQ AX, AX -MOVQ s+0(FP), SI -MOVQ s_len+8(FP), CX -TESTQ CX, CX -JZ popcntXorSliceEnd -MOVQ m+24(FP), DI -popcntXorSliceLoop: -MOVQ (DI), DX -XORQ (SI), DX -POPCNTQ_DX_DX -ADDQ DX, AX -ADDQ $8, SI -ADDQ $8, DI -LOOP popcntXorSliceLoop -popcntXorSliceEnd: -MOVQ AX, ret+48(FP) -RET diff --git a/vendor/github.com/willf/bitset/popcnt_generic.go b/vendor/github.com/willf/bitset/popcnt_generic.go deleted file mode 100644 index 6b21cb7..0000000 --- a/vendor/github.com/willf/bitset/popcnt_generic.go +++ /dev/null @@ -1,23 +0,0 @@ -// +build !amd64 appengine - -package bitset - -func popcntSlice(s []uint64) uint64 { - return popcntSliceGo(s) -} - -func popcntMaskSlice(s, m []uint64) uint64 { - return popcntMaskSliceGo(s, m) -} - -func popcntAndSlice(s, m []uint64) uint64 { - return popcntAndSliceGo(s, m) -} - -func popcntOrSlice(s, m []uint64) uint64 { - return popcntOrSliceGo(s, m) -} - -func popcntXorSlice(s, m []uint64) uint64 { - return popcntXorSliceGo(s, m) -} diff --git a/vendor/golang.org/x/crypto/sha3/keccakf.go b/vendor/golang.org/x/crypto/sha3/keccakf.go index 13e7058..46d03ed 100644 --- a/vendor/golang.org/x/crypto/sha3/keccakf.go +++ b/vendor/golang.org/x/crypto/sha3/keccakf.go @@ -2,6 +2,8 @@ // Use of this source code is governed by a BSD-style // license that can be found in the LICENSE file. +// +build !amd64 appengine gccgo + package sha3 // rc stores the round constants for use in the ι step. diff --git a/vendor/golang.org/x/crypto/sha3/keccakf_amd64.go b/vendor/golang.org/x/crypto/sha3/keccakf_amd64.go new file mode 100644 index 0000000..7886795 --- /dev/null +++ b/vendor/golang.org/x/crypto/sha3/keccakf_amd64.go @@ -0,0 +1,13 @@ +// Copyright 2015 The Go Authors. All rights reserved. +// Use of this source code is governed by a BSD-style +// license that can be found in the LICENSE file. + +// +build amd64,!appengine,!gccgo + +package sha3 + +// This function is implemented in keccakf_amd64.s. + +//go:noescape + +func keccakF1600(a *[25]uint64) diff --git a/vendor/golang.org/x/crypto/sha3/keccakf_amd64.s b/vendor/golang.org/x/crypto/sha3/keccakf_amd64.s new file mode 100644 index 0000000..f88533a --- /dev/null +++ b/vendor/golang.org/x/crypto/sha3/keccakf_amd64.s @@ -0,0 +1,390 @@ +// Copyright 2015 The Go Authors. All rights reserved. +// Use of this source code is governed by a BSD-style +// license that can be found in the LICENSE file. + +// +build amd64,!appengine,!gccgo + +// This code was translated into a form compatible with 6a from the public +// domain sources at https://github.com/gvanas/KeccakCodePackage + +// Offsets in state +#define _ba (0*8) +#define _be (1*8) +#define _bi (2*8) +#define _bo (3*8) +#define _bu (4*8) +#define _ga (5*8) +#define _ge (6*8) +#define _gi (7*8) +#define _go (8*8) +#define _gu (9*8) +#define _ka (10*8) +#define _ke (11*8) +#define _ki (12*8) +#define _ko (13*8) +#define _ku (14*8) +#define _ma (15*8) +#define _me (16*8) +#define _mi (17*8) +#define _mo (18*8) +#define _mu (19*8) +#define _sa (20*8) +#define _se (21*8) +#define _si (22*8) +#define _so (23*8) +#define _su (24*8) + +// Temporary registers +#define rT1 AX + +// Round vars +#define rpState DI +#define rpStack SP + +#define rDa BX +#define rDe CX +#define rDi DX +#define rDo R8 +#define rDu R9 + +#define rBa R10 +#define rBe R11 +#define rBi R12 +#define rBo R13 +#define rBu R14 + +#define rCa SI +#define rCe BP +#define rCi rBi +#define rCo rBo +#define rCu R15 + +#define MOVQ_RBI_RCE MOVQ rBi, rCe +#define XORQ_RT1_RCA XORQ rT1, rCa +#define XORQ_RT1_RCE XORQ rT1, rCe +#define XORQ_RBA_RCU XORQ rBa, rCu +#define XORQ_RBE_RCU XORQ rBe, rCu +#define XORQ_RDU_RCU XORQ rDu, rCu +#define XORQ_RDA_RCA XORQ rDa, rCa +#define XORQ_RDE_RCE XORQ rDe, rCe + +#define mKeccakRound(iState, oState, rc, B_RBI_RCE, G_RT1_RCA, G_RT1_RCE, G_RBA_RCU, K_RT1_RCA, K_RT1_RCE, K_RBA_RCU, M_RT1_RCA, M_RT1_RCE, M_RBE_RCU, S_RDU_RCU, S_RDA_RCA, S_RDE_RCE) \ + /* Prepare round */ \ + MOVQ rCe, rDa; \ + ROLQ $1, rDa; \ + \ + MOVQ _bi(iState), rCi; \ + XORQ _gi(iState), rDi; \ + XORQ rCu, rDa; \ + XORQ _ki(iState), rCi; \ + XORQ _mi(iState), rDi; \ + XORQ rDi, rCi; \ + \ + MOVQ rCi, rDe; \ + ROLQ $1, rDe; \ + \ + MOVQ _bo(iState), rCo; \ + XORQ _go(iState), rDo; \ + XORQ rCa, rDe; \ + XORQ _ko(iState), rCo; \ + XORQ _mo(iState), rDo; \ + XORQ rDo, rCo; \ + \ + MOVQ rCo, rDi; \ + ROLQ $1, rDi; \ + \ + MOVQ rCu, rDo; \ + XORQ rCe, rDi; \ + ROLQ $1, rDo; \ + \ + MOVQ rCa, rDu; \ + XORQ rCi, rDo; \ + ROLQ $1, rDu; \ + \ + /* Result b */ \ + MOVQ _ba(iState), rBa; \ + MOVQ _ge(iState), rBe; \ + XORQ rCo, rDu; \ + MOVQ _ki(iState), rBi; \ + MOVQ _mo(iState), rBo; \ + MOVQ _su(iState), rBu; \ + XORQ rDe, rBe; \ + ROLQ $44, rBe; \ + XORQ rDi, rBi; \ + XORQ rDa, rBa; \ + ROLQ $43, rBi; \ + \ + MOVQ rBe, rCa; \ + MOVQ rc, rT1; \ + ORQ rBi, rCa; \ + XORQ rBa, rT1; \ + XORQ rT1, rCa; \ + MOVQ rCa, _ba(oState); \ + \ + XORQ rDu, rBu; \ + ROLQ $14, rBu; \ + MOVQ rBa, rCu; \ + ANDQ rBe, rCu; \ + XORQ rBu, rCu; \ + MOVQ rCu, _bu(oState); \ + \ + XORQ rDo, rBo; \ + ROLQ $21, rBo; \ + MOVQ rBo, rT1; \ + ANDQ rBu, rT1; \ + XORQ rBi, rT1; \ + MOVQ rT1, _bi(oState); \ + \ + NOTQ rBi; \ + ORQ rBa, rBu; \ + ORQ rBo, rBi; \ + XORQ rBo, rBu; \ + XORQ rBe, rBi; \ + MOVQ rBu, _bo(oState); \ + MOVQ rBi, _be(oState); \ + B_RBI_RCE; \ + \ + /* Result g */ \ + MOVQ _gu(iState), rBe; \ + XORQ rDu, rBe; \ + MOVQ _ka(iState), rBi; \ + ROLQ $20, rBe; \ + XORQ rDa, rBi; \ + ROLQ $3, rBi; \ + MOVQ _bo(iState), rBa; \ + MOVQ rBe, rT1; \ + ORQ rBi, rT1; \ + XORQ rDo, rBa; \ + MOVQ _me(iState), rBo; \ + MOVQ _si(iState), rBu; \ + ROLQ $28, rBa; \ + XORQ rBa, rT1; \ + MOVQ rT1, _ga(oState); \ + G_RT1_RCA; \ + \ + XORQ rDe, rBo; \ + ROLQ $45, rBo; \ + MOVQ rBi, rT1; \ + ANDQ rBo, rT1; \ + XORQ rBe, rT1; \ + MOVQ rT1, _ge(oState); \ + G_RT1_RCE; \ + \ + XORQ rDi, rBu; \ + ROLQ $61, rBu; \ + MOVQ rBu, rT1; \ + ORQ rBa, rT1; \ + XORQ rBo, rT1; \ + MOVQ rT1, _go(oState); \ + \ + ANDQ rBe, rBa; \ + XORQ rBu, rBa; \ + MOVQ rBa, _gu(oState); \ + NOTQ rBu; \ + G_RBA_RCU; \ + \ + ORQ rBu, rBo; \ + XORQ rBi, rBo; \ + MOVQ rBo, _gi(oState); \ + \ + /* Result k */ \ + MOVQ _be(iState), rBa; \ + MOVQ _gi(iState), rBe; \ + MOVQ _ko(iState), rBi; \ + MOVQ _mu(iState), rBo; \ + MOVQ _sa(iState), rBu; \ + XORQ rDi, rBe; \ + ROLQ $6, rBe; \ + XORQ rDo, rBi; \ + ROLQ $25, rBi; \ + MOVQ rBe, rT1; \ + ORQ rBi, rT1; \ + XORQ rDe, rBa; \ + ROLQ $1, rBa; \ + XORQ rBa, rT1; \ + MOVQ rT1, _ka(oState); \ + K_RT1_RCA; \ + \ + XORQ rDu, rBo; \ + ROLQ $8, rBo; \ + MOVQ rBi, rT1; \ + ANDQ rBo, rT1; \ + XORQ rBe, rT1; \ + MOVQ rT1, _ke(oState); \ + K_RT1_RCE; \ + \ + XORQ rDa, rBu; \ + ROLQ $18, rBu; \ + NOTQ rBo; \ + MOVQ rBo, rT1; \ + ANDQ rBu, rT1; \ + XORQ rBi, rT1; \ + MOVQ rT1, _ki(oState); \ + \ + MOVQ rBu, rT1; \ + ORQ rBa, rT1; \ + XORQ rBo, rT1; \ + MOVQ rT1, _ko(oState); \ + \ + ANDQ rBe, rBa; \ + XORQ rBu, rBa; \ + MOVQ rBa, _ku(oState); \ + K_RBA_RCU; \ + \ + /* Result m */ \ + MOVQ _ga(iState), rBe; \ + XORQ rDa, rBe; \ + MOVQ _ke(iState), rBi; \ + ROLQ $36, rBe; \ + XORQ rDe, rBi; \ + MOVQ _bu(iState), rBa; \ + ROLQ $10, rBi; \ + MOVQ rBe, rT1; \ + MOVQ _mi(iState), rBo; \ + ANDQ rBi, rT1; \ + XORQ rDu, rBa; \ + MOVQ _so(iState), rBu; \ + ROLQ $27, rBa; \ + XORQ rBa, rT1; \ + MOVQ rT1, _ma(oState); \ + M_RT1_RCA; \ + \ + XORQ rDi, rBo; \ + ROLQ $15, rBo; \ + MOVQ rBi, rT1; \ + ORQ rBo, rT1; \ + XORQ rBe, rT1; \ + MOVQ rT1, _me(oState); \ + M_RT1_RCE; \ + \ + XORQ rDo, rBu; \ + ROLQ $56, rBu; \ + NOTQ rBo; \ + MOVQ rBo, rT1; \ + ORQ rBu, rT1; \ + XORQ rBi, rT1; \ + MOVQ rT1, _mi(oState); \ + \ + ORQ rBa, rBe; \ + XORQ rBu, rBe; \ + MOVQ rBe, _mu(oState); \ + \ + ANDQ rBa, rBu; \ + XORQ rBo, rBu; \ + MOVQ rBu, _mo(oState); \ + M_RBE_RCU; \ + \ + /* Result s */ \ + MOVQ _bi(iState), rBa; \ + MOVQ _go(iState), rBe; \ + MOVQ _ku(iState), rBi; \ + XORQ rDi, rBa; \ + MOVQ _ma(iState), rBo; \ + ROLQ $62, rBa; \ + XORQ rDo, rBe; \ + MOVQ _se(iState), rBu; \ + ROLQ $55, rBe; \ + \ + XORQ rDu, rBi; \ + MOVQ rBa, rDu; \ + XORQ rDe, rBu; \ + ROLQ $2, rBu; \ + ANDQ rBe, rDu; \ + XORQ rBu, rDu; \ + MOVQ rDu, _su(oState); \ + \ + ROLQ $39, rBi; \ + S_RDU_RCU; \ + NOTQ rBe; \ + XORQ rDa, rBo; \ + MOVQ rBe, rDa; \ + ANDQ rBi, rDa; \ + XORQ rBa, rDa; \ + MOVQ rDa, _sa(oState); \ + S_RDA_RCA; \ + \ + ROLQ $41, rBo; \ + MOVQ rBi, rDe; \ + ORQ rBo, rDe; \ + XORQ rBe, rDe; \ + MOVQ rDe, _se(oState); \ + S_RDE_RCE; \ + \ + MOVQ rBo, rDi; \ + MOVQ rBu, rDo; \ + ANDQ rBu, rDi; \ + ORQ rBa, rDo; \ + XORQ rBi, rDi; \ + XORQ rBo, rDo; \ + MOVQ rDi, _si(oState); \ + MOVQ rDo, _so(oState) \ + +// func keccakF1600(state *[25]uint64) +TEXT ·keccakF1600(SB), 0, $200-8 + MOVQ state+0(FP), rpState + + // Convert the user state into an internal state + NOTQ _be(rpState) + NOTQ _bi(rpState) + NOTQ _go(rpState) + NOTQ _ki(rpState) + NOTQ _mi(rpState) + NOTQ _sa(rpState) + + // Execute the KeccakF permutation + MOVQ _ba(rpState), rCa + MOVQ _be(rpState), rCe + MOVQ _bu(rpState), rCu + + XORQ _ga(rpState), rCa + XORQ _ge(rpState), rCe + XORQ _gu(rpState), rCu + + XORQ _ka(rpState), rCa + XORQ _ke(rpState), rCe + XORQ _ku(rpState), rCu + + XORQ _ma(rpState), rCa + XORQ _me(rpState), rCe + XORQ _mu(rpState), rCu + + XORQ _sa(rpState), rCa + XORQ _se(rpState), rCe + MOVQ _si(rpState), rDi + MOVQ _so(rpState), rDo + XORQ _su(rpState), rCu + + mKeccakRound(rpState, rpStack, $0x0000000000000001, MOVQ_RBI_RCE, XORQ_RT1_RCA, XORQ_RT1_RCE, XORQ_RBA_RCU, XORQ_RT1_RCA, XORQ_RT1_RCE, XORQ_RBA_RCU, XORQ_RT1_RCA, XORQ_RT1_RCE, XORQ_RBE_RCU, XORQ_RDU_RCU, XORQ_RDA_RCA, XORQ_RDE_RCE) + mKeccakRound(rpStack, rpState, $0x0000000000008082, MOVQ_RBI_RCE, XORQ_RT1_RCA, XORQ_RT1_RCE, XORQ_RBA_RCU, XORQ_RT1_RCA, XORQ_RT1_RCE, XORQ_RBA_RCU, XORQ_RT1_RCA, XORQ_RT1_RCE, XORQ_RBE_RCU, XORQ_RDU_RCU, XORQ_RDA_RCA, XORQ_RDE_RCE) + mKeccakRound(rpState, rpStack, $0x800000000000808a, MOVQ_RBI_RCE, XORQ_RT1_RCA, XORQ_RT1_RCE, XORQ_RBA_RCU, XORQ_RT1_RCA, XORQ_RT1_RCE, XORQ_RBA_RCU, XORQ_RT1_RCA, XORQ_RT1_RCE, XORQ_RBE_RCU, XORQ_RDU_RCU, XORQ_RDA_RCA, XORQ_RDE_RCE) + mKeccakRound(rpStack, rpState, $0x8000000080008000, MOVQ_RBI_RCE, XORQ_RT1_RCA, XORQ_RT1_RCE, XORQ_RBA_RCU, XORQ_RT1_RCA, XORQ_RT1_RCE, XORQ_RBA_RCU, XORQ_RT1_RCA, XORQ_RT1_RCE, XORQ_RBE_RCU, XORQ_RDU_RCU, XORQ_RDA_RCA, XORQ_RDE_RCE) + mKeccakRound(rpState, rpStack, $0x000000000000808b, MOVQ_RBI_RCE, XORQ_RT1_RCA, XORQ_RT1_RCE, XORQ_RBA_RCU, XORQ_RT1_RCA, XORQ_RT1_RCE, XORQ_RBA_RCU, XORQ_RT1_RCA, XORQ_RT1_RCE, XORQ_RBE_RCU, XORQ_RDU_RCU, XORQ_RDA_RCA, XORQ_RDE_RCE) + mKeccakRound(rpStack, rpState, $0x0000000080000001, MOVQ_RBI_RCE, XORQ_RT1_RCA, XORQ_RT1_RCE, XORQ_RBA_RCU, XORQ_RT1_RCA, XORQ_RT1_RCE, XORQ_RBA_RCU, XORQ_RT1_RCA, XORQ_RT1_RCE, XORQ_RBE_RCU, XORQ_RDU_RCU, XORQ_RDA_RCA, XORQ_RDE_RCE) + mKeccakRound(rpState, rpStack, $0x8000000080008081, MOVQ_RBI_RCE, XORQ_RT1_RCA, XORQ_RT1_RCE, XORQ_RBA_RCU, XORQ_RT1_RCA, XORQ_RT1_RCE, XORQ_RBA_RCU, XORQ_RT1_RCA, XORQ_RT1_RCE, XORQ_RBE_RCU, XORQ_RDU_RCU, XORQ_RDA_RCA, XORQ_RDE_RCE) + mKeccakRound(rpStack, rpState, $0x8000000000008009, MOVQ_RBI_RCE, XORQ_RT1_RCA, XORQ_RT1_RCE, XORQ_RBA_RCU, XORQ_RT1_RCA, XORQ_RT1_RCE, XORQ_RBA_RCU, XORQ_RT1_RCA, XORQ_RT1_RCE, XORQ_RBE_RCU, XORQ_RDU_RCU, XORQ_RDA_RCA, XORQ_RDE_RCE) + mKeccakRound(rpState, rpStack, $0x000000000000008a, MOVQ_RBI_RCE, XORQ_RT1_RCA, XORQ_RT1_RCE, XORQ_RBA_RCU, XORQ_RT1_RCA, XORQ_RT1_RCE, XORQ_RBA_RCU, XORQ_RT1_RCA, XORQ_RT1_RCE, XORQ_RBE_RCU, XORQ_RDU_RCU, XORQ_RDA_RCA, XORQ_RDE_RCE) + mKeccakRound(rpStack, rpState, $0x0000000000000088, MOVQ_RBI_RCE, XORQ_RT1_RCA, XORQ_RT1_RCE, XORQ_RBA_RCU, XORQ_RT1_RCA, XORQ_RT1_RCE, XORQ_RBA_RCU, XORQ_RT1_RCA, XORQ_RT1_RCE, XORQ_RBE_RCU, XORQ_RDU_RCU, XORQ_RDA_RCA, XORQ_RDE_RCE) + mKeccakRound(rpState, rpStack, $0x0000000080008009, MOVQ_RBI_RCE, XORQ_RT1_RCA, XORQ_RT1_RCE, XORQ_RBA_RCU, XORQ_RT1_RCA, XORQ_RT1_RCE, XORQ_RBA_RCU, XORQ_RT1_RCA, XORQ_RT1_RCE, XORQ_RBE_RCU, XORQ_RDU_RCU, XORQ_RDA_RCA, XORQ_RDE_RCE) + mKeccakRound(rpStack, rpState, $0x000000008000000a, MOVQ_RBI_RCE, XORQ_RT1_RCA, XORQ_RT1_RCE, XORQ_RBA_RCU, XORQ_RT1_RCA, XORQ_RT1_RCE, XORQ_RBA_RCU, XORQ_RT1_RCA, XORQ_RT1_RCE, XORQ_RBE_RCU, XORQ_RDU_RCU, XORQ_RDA_RCA, XORQ_RDE_RCE) + mKeccakRound(rpState, rpStack, $0x000000008000808b, MOVQ_RBI_RCE, XORQ_RT1_RCA, XORQ_RT1_RCE, XORQ_RBA_RCU, XORQ_RT1_RCA, XORQ_RT1_RCE, XORQ_RBA_RCU, XORQ_RT1_RCA, XORQ_RT1_RCE, XORQ_RBE_RCU, XORQ_RDU_RCU, XORQ_RDA_RCA, XORQ_RDE_RCE) + mKeccakRound(rpStack, rpState, $0x800000000000008b, MOVQ_RBI_RCE, XORQ_RT1_RCA, XORQ_RT1_RCE, XORQ_RBA_RCU, XORQ_RT1_RCA, XORQ_RT1_RCE, XORQ_RBA_RCU, XORQ_RT1_RCA, XORQ_RT1_RCE, XORQ_RBE_RCU, XORQ_RDU_RCU, XORQ_RDA_RCA, XORQ_RDE_RCE) + mKeccakRound(rpState, rpStack, $0x8000000000008089, MOVQ_RBI_RCE, XORQ_RT1_RCA, XORQ_RT1_RCE, XORQ_RBA_RCU, XORQ_RT1_RCA, XORQ_RT1_RCE, XORQ_RBA_RCU, XORQ_RT1_RCA, XORQ_RT1_RCE, XORQ_RBE_RCU, XORQ_RDU_RCU, XORQ_RDA_RCA, XORQ_RDE_RCE) + mKeccakRound(rpStack, rpState, $0x8000000000008003, MOVQ_RBI_RCE, XORQ_RT1_RCA, XORQ_RT1_RCE, XORQ_RBA_RCU, XORQ_RT1_RCA, XORQ_RT1_RCE, XORQ_RBA_RCU, XORQ_RT1_RCA, XORQ_RT1_RCE, XORQ_RBE_RCU, XORQ_RDU_RCU, XORQ_RDA_RCA, XORQ_RDE_RCE) + mKeccakRound(rpState, rpStack, $0x8000000000008002, MOVQ_RBI_RCE, XORQ_RT1_RCA, XORQ_RT1_RCE, XORQ_RBA_RCU, XORQ_RT1_RCA, XORQ_RT1_RCE, XORQ_RBA_RCU, XORQ_RT1_RCA, XORQ_RT1_RCE, XORQ_RBE_RCU, XORQ_RDU_RCU, XORQ_RDA_RCA, XORQ_RDE_RCE) + mKeccakRound(rpStack, rpState, $0x8000000000000080, MOVQ_RBI_RCE, XORQ_RT1_RCA, XORQ_RT1_RCE, XORQ_RBA_RCU, XORQ_RT1_RCA, XORQ_RT1_RCE, XORQ_RBA_RCU, XORQ_RT1_RCA, XORQ_RT1_RCE, XORQ_RBE_RCU, XORQ_RDU_RCU, XORQ_RDA_RCA, XORQ_RDE_RCE) + mKeccakRound(rpState, rpStack, $0x000000000000800a, MOVQ_RBI_RCE, XORQ_RT1_RCA, XORQ_RT1_RCE, XORQ_RBA_RCU, XORQ_RT1_RCA, XORQ_RT1_RCE, XORQ_RBA_RCU, XORQ_RT1_RCA, XORQ_RT1_RCE, XORQ_RBE_RCU, XORQ_RDU_RCU, XORQ_RDA_RCA, XORQ_RDE_RCE) + mKeccakRound(rpStack, rpState, $0x800000008000000a, MOVQ_RBI_RCE, XORQ_RT1_RCA, XORQ_RT1_RCE, XORQ_RBA_RCU, XORQ_RT1_RCA, XORQ_RT1_RCE, XORQ_RBA_RCU, XORQ_RT1_RCA, XORQ_RT1_RCE, XORQ_RBE_RCU, XORQ_RDU_RCU, XORQ_RDA_RCA, XORQ_RDE_RCE) + mKeccakRound(rpState, rpStack, $0x8000000080008081, MOVQ_RBI_RCE, XORQ_RT1_RCA, XORQ_RT1_RCE, XORQ_RBA_RCU, XORQ_RT1_RCA, XORQ_RT1_RCE, XORQ_RBA_RCU, XORQ_RT1_RCA, XORQ_RT1_RCE, XORQ_RBE_RCU, XORQ_RDU_RCU, XORQ_RDA_RCA, XORQ_RDE_RCE) + mKeccakRound(rpStack, rpState, $0x8000000000008080, MOVQ_RBI_RCE, XORQ_RT1_RCA, XORQ_RT1_RCE, XORQ_RBA_RCU, XORQ_RT1_RCA, XORQ_RT1_RCE, XORQ_RBA_RCU, XORQ_RT1_RCA, XORQ_RT1_RCE, XORQ_RBE_RCU, XORQ_RDU_RCU, XORQ_RDA_RCA, XORQ_RDE_RCE) + mKeccakRound(rpState, rpStack, $0x0000000080000001, MOVQ_RBI_RCE, XORQ_RT1_RCA, XORQ_RT1_RCE, XORQ_RBA_RCU, XORQ_RT1_RCA, XORQ_RT1_RCE, XORQ_RBA_RCU, XORQ_RT1_RCA, XORQ_RT1_RCE, XORQ_RBE_RCU, XORQ_RDU_RCU, XORQ_RDA_RCA, XORQ_RDE_RCE) + mKeccakRound(rpStack, rpState, $0x8000000080008008, NOP, NOP, NOP, NOP, NOP, NOP, NOP, NOP, NOP, NOP, NOP, NOP, NOP) + + // Revert the internal state to the user state + NOTQ _be(rpState) + NOTQ _bi(rpState) + NOTQ _go(rpState) + NOTQ _ki(rpState) + NOTQ _mi(rpState) + NOTQ _sa(rpState) + + RET diff --git a/vendor/golang.org/x/crypto/sha3/sha3.go b/vendor/golang.org/x/crypto/sha3/sha3.go index c8fd31c..c86167c 100644 --- a/vendor/golang.org/x/crypto/sha3/sha3.go +++ b/vendor/golang.org/x/crypto/sha3/sha3.go @@ -42,7 +42,7 @@ type state struct { storage [maxRate]byte // Specific to SHA-3 and SHAKE. - fixedOutput bool // whether this is a fixed-ouput-length instance + fixedOutput bool // whether this is a fixed-output-length instance outputLen int // the default output size in bytes state spongeDirection // whether the sponge is absorbing or squeezing }