Initial dictionary producer: builddict, word-list sources, DAWG build + CI
build / dawg (push) Failing after 2s
build / dawg (push) Failing after 2s
- builddict drives the de-internalized scrabble-solver dictdawg/wordlist builders (pinned v1.0.0) to produce the three DAWGs (en_sowpods, ru_scrabble, ru_erudit), byte-identical to the solver's committed fixtures (same dafsa/alphabet v1.1.0 -> no index drift with the running backend). - Sources: english/sowpods.txt vendored from kamilmielnik/scrabble-dictionaries; russian/scrabble.txt + the dictprep tooling moved out of scrabble-solver. - CI builds the DAWGs on push/PR and, on a vX.Y.Z tag, packages them flat into scrabble-dawg-<tag>.tar.gz and attaches it to the Gitea release.
This commit is contained in:
@@ -0,0 +1,75 @@
|
||||
// Command builddict converts a word list into a serialized DAWG. By default it reads the
|
||||
// English SOWPODS list (Latin alphabet); pass -alphabet russian for the Cyrillic lists.
|
||||
package main
|
||||
|
||||
import (
|
||||
"flag"
|
||||
"fmt"
|
||||
"log"
|
||||
"os"
|
||||
"path/filepath"
|
||||
"time"
|
||||
|
||||
"github.com/iliadenisov/alphabet"
|
||||
|
||||
"gitea.iliadenisov.ru/developer/scrabble-solver/dictdawg"
|
||||
"gitea.iliadenisov.ru/developer/scrabble-solver/wordlist"
|
||||
)
|
||||
|
||||
func main() {
|
||||
dict := flag.String("dict", "dictionaries/english/sowpods.txt", "word list file (one word per line)")
|
||||
out := flag.String("out", "testdata", "output directory")
|
||||
name := flag.String("name", "sowpods", "base name for the output file")
|
||||
minLen := flag.Int("min", 2, "minimum word length")
|
||||
maxLen := flag.Int("max", 15, "maximum word length")
|
||||
alpha := flag.String("alphabet", "latin", "alphabet: latin (English) or russian")
|
||||
flag.Parse()
|
||||
|
||||
var idx alphabet.Indexer
|
||||
switch *alpha {
|
||||
case "latin":
|
||||
idx = alphabet.Latin()
|
||||
case "russian":
|
||||
idx = alphabet.Embedded(alphabet.Langs.LangRu)
|
||||
default:
|
||||
log.Fatalf("unknown -alphabet %q (want latin or russian)", *alpha)
|
||||
}
|
||||
|
||||
t0 := time.Now()
|
||||
words, err := wordlist.Read(*dict, idx, *minLen, *maxLen)
|
||||
if err != nil {
|
||||
log.Fatalf("read %s: %v", *dict, err)
|
||||
}
|
||||
fmt.Printf("loaded %d words from %s in %s\n", len(words), *dict, time.Since(t0).Round(time.Millisecond))
|
||||
|
||||
if err := os.MkdirAll(*out, 0o755); err != nil {
|
||||
log.Fatal(err)
|
||||
}
|
||||
|
||||
t := time.Now()
|
||||
f, err := dictdawg.Build(idx, words)
|
||||
if err != nil {
|
||||
log.Fatalf("build dawg: %v", err)
|
||||
}
|
||||
path := filepath.Join(*out, *name+".dawg")
|
||||
if err := dictdawg.Save(f, path); err != nil {
|
||||
log.Fatalf("save: %v", err)
|
||||
}
|
||||
size := int64(0)
|
||||
if fi, err := os.Stat(path); err == nil {
|
||||
size = fi.Size()
|
||||
}
|
||||
fmt.Printf("DAWG %d nodes, %s, built+saved in %s -> %s\n",
|
||||
f.NumNodes(), humanBytes(size), time.Since(t).Round(time.Millisecond), path)
|
||||
}
|
||||
|
||||
func humanBytes(n int64) string {
|
||||
switch {
|
||||
case n >= 1<<20:
|
||||
return fmt.Sprintf("%.2f MB", float64(n)/(1<<20))
|
||||
case n >= 1<<10:
|
||||
return fmt.Sprintf("%.1f KB", float64(n)/(1<<10))
|
||||
default:
|
||||
return fmt.Sprintf("%d B", n)
|
||||
}
|
||||
}
|
||||
Reference in New Issue
Block a user