[package] name = "tokenizers" description = "Text tokenizers for pg_search" version = { workspace = true } edition = { workspace = true } license = { workspace = true } [lints] workspace = true [dependencies] anyhow = "1.0.102" # The segmenter and the dictionaries live in `lindera`; the character and token # filter chain we layer on top lives in `lindera-analysis`. lindera = { version = "5.0.1", features = [ "embed-cc-cedict", "embed-ipadic", "embed-ko-dic", ] } lindera-analysis = "5.0.1" once_cell = "1.21.4" serde = "1.0.228" serde_json = "1.0.149" tantivy.workspace = true tracing = "0.1.44" strum_macros = "0.28.0" strum = { version = "0.28.0", features = ["derive"] } tantivy-jieba = { workspace = true } emojis = "0.9.0" icu_properties = "2.2.0" unicode-segmentation = "1.13.2" icu_segmenter = "2.2.0" # We use a fork of opencc-jieba-rs to remove the exact-version dependency pins (e.g., include-flate = "=0.3.0") # that were introduced for MSRV reasons, which otherwise conflict with our newer dependencies. # The fork also bumps jieba-rs to 0.10, matching tantivy-jieba so we compile one jieba chain rather than two. opencc-jieba-rs = { git = "https://github.com/paradedb/opencc-jieba-rs", branch = "main" } [dev-dependencies] rstest = "0.26.1" [package.metadata.cargo-machete] ignored = ["strum"]