Merge pull request #4195 from vespa-engine/bratseth/token-symmetry

Test tokens with accents are processed symmetrically
vespa-engine · Sep 11, 2024 · 024d379 · 024d379
2 parents 0e7b3c1 + 8498307
commit 024d379
Show file tree

Hide file tree

Showing 2 changed files with 4 additions and 2 deletions.
diff --git a/tests/search/linguistics/open-nlp/documents.json b/tests/search/linguistics/open-nlp/documents.json
@@ -1,4 +1,5 @@
 [
   {"id":"id:test:test::doc1","fields":{"text":"是一个展示雅，目前在测试阶段。"}},
-  {"id":"id:test:test::doc2","fields":{"text":"Will still be stemmed: Cars."}}
+  {"id":"id:test:test::doc2","fields":{"text":"Will still be stemmed: Cars."}},
+  {"id":"id:test:test::doc3","fields":{"text":"congés"}}
 ]
diff --git a/tests/search/linguistics/open-nlp/opennlp_linguistics.rb b/tests/search/linguistics/open-nlp/opennlp_linguistics.rb
@@ -5,7 +5,7 @@ class OpenNlpLinguistics < SearchTest
 
   def setup
     set_owner("bratseth")
-    set_description("Tests Chinese segmentation with the OpenNlp linguistics module")
+    set_description("Tests the OpenNlp linguistics module")
   end
 
   def make_app
@@ -31,6 +31,7 @@ def test_opennlp_linguistics
 
     assert_hitcount("query=text:展示", 1) # A Chinese token from the resulting segmentation done
     assert_hitcount("query=text:car", 1) # English is still stemmed
+    assert_hitcount("query=text:congés", 1)
    end
 
   def teardown