tesseract/unittest/unichar_test.cc

// (C) Copyright 2017, Google Inc.
// Licensed under the Apache License, Version 2.0 (the "License");
// you may not use this file except in compliance with the License.
// You may obtain a copy of the License at
// http://www.apache.org/licenses/LICENSE-2.0
// Unless required by applicable law or agreed to in writing, software
// distributed under the License is distributed on an "AS IS" BASIS,
// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
// See the License for the specific language governing permissions and
// limitations under the License.

#include "include_gunit.h"
#include "gmock/gmock.h"                // for testing::ElementsAreArray
#include <tesseract/unichar.h>

namespace tesseract {

TEST(UnicharTest, Conversion) {
  // This test verifies that Unichar::UTF8ToUTF32 and Unichar::UTF32ToUTF8
  // show the required conversion properties.
  // Test for round-trip utf8-32-8 for 1, 2, 3 and 4 byte codes.
  const char* kUTF8Src = "a\u05d0\u0ca4\U0002a714";
  const std::vector<char32> kUTF32Src = {'a', 0x5d0, 0xca4, 0x2a714};
  // Check for round-trip conversion.
  std::vector<char32> utf32 = UNICHAR::UTF8ToUTF32(kUTF8Src);
  EXPECT_THAT(utf32, testing::ElementsAreArray(kUTF32Src));
  std::string utf8 = UNICHAR::UTF32ToUTF8(utf32);
  EXPECT_STREQ(kUTF8Src, utf8.c_str());
}

TEST(UnicharTest, InvalidText) {
  // This test verifies that Unichar correctly deals with invalid text.
  const char* kInvalidUTF8 = "a b\200d string";
  const std::vector<char32> kInvalidUTF32 = {'a', ' ', 0x200000, 'x'};
  // Invalid utf8 produces an empty vector.
  std::vector<char32> utf32 = UNICHAR::UTF8ToUTF32(kInvalidUTF8);
  EXPECT_TRUE(utf32.empty());
  // Invalid utf32 produces an empty string.
  std::string utf8 = UNICHAR::UTF32ToUTF8(kInvalidUTF32);
  EXPECT_TRUE(utf8.empty());
}

}  // namespace
Fix unicharset_test 2019-01-19 00:41:29 +08:00			`// (C) Copyright 2017, Google Inc.`
			`// Licensed under the Apache License, Version 2.0 (the "License");`
			`// you may not use this file except in compliance with the License.`
			`// You may obtain a copy of the License at`
			`// http://www.apache.org/licenses/LICENSE-2.0`
			`// Unless required by applicable law or agreed to in writing, software`
			`// distributed under the License is distributed on an "AS IS" BASIS,`
			`// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.`
			`// See the License for the specific language governing permissions and`
			`// limitations under the License.`

			`#include "include_gunit.h"`
Fix build for unichar_test Signed-off-by: Stefan Weil <sw@weilnetz.de> 2019-01-19 04:15:16 +08:00			`#include "gmock/gmock.h" // for testing::ElementsAreArray`
Use #include <tesseract/*.h> for unittest Signed-off-by: Stefan Weil <sw@weilnetz.de> 2019-10-30 01:01:18 +08:00			`#include <tesseract/unichar.h>`
Add more unittests from Google They were provided by Jeff Breidenbach <jbreiden@google.com>. Signed-off-by: Stefan Weil <sw@weilnetz.de> 2018-08-24 21:07:48 +08:00
Add / fix namespace tesseract for unittest Signed-off-by: Stefan Weil <sw@weilnetz.de> 2020-12-27 17:41:48 +08:00			`namespace tesseract {`
Add more unittests from Google They were provided by Jeff Breidenbach <jbreiden@google.com>. Signed-off-by: Stefan Weil <sw@weilnetz.de> 2018-08-24 21:07:48 +08:00
			`TEST(UnicharTest, Conversion) {`
			`// This test verifies that Unichar::UTF8ToUTF32 and Unichar::UTF32ToUTF8`
			`// show the required conversion properties.`
			`// Test for round-trip utf8-32-8 for 1, 2, 3 and 4 byte codes.`
			`const char* kUTF8Src = "a\u05d0\u0ca4\U0002a714";`
			`const std::vector<char32> kUTF32Src = {'a', 0x5d0, 0xca4, 0x2a714};`
			`// Check for round-trip conversion.`
			`std::vector<char32> utf32 = UNICHAR::UTF8ToUTF32(kUTF8Src);`
			`EXPECT_THAT(utf32, testing::ElementsAreArray(kUTF32Src));`
Fix unicharset_test 2019-01-19 00:41:29 +08:00			`std::string utf8 = UNICHAR::UTF32ToUTF8(utf32);`
Add more unittests from Google They were provided by Jeff Breidenbach <jbreiden@google.com>. Signed-off-by: Stefan Weil <sw@weilnetz.de> 2018-08-24 21:07:48 +08:00			`EXPECT_STREQ(kUTF8Src, utf8.c_str());`
			`}`

			`TEST(UnicharTest, InvalidText) {`
			`// This test verifies that Unichar correctly deals with invalid text.`
			`const char* kInvalidUTF8 = "a b\200d string";`
			`const std::vector<char32> kInvalidUTF32 = {'a', ' ', 0x200000, 'x'};`
			`// Invalid utf8 produces an empty vector.`
			`std::vector<char32> utf32 = UNICHAR::UTF8ToUTF32(kInvalidUTF8);`
			`EXPECT_TRUE(utf32.empty());`
			`// Invalid utf32 produces an empty string.`
Fix unicharset_test 2019-01-19 00:41:29 +08:00			`std::string utf8 = UNICHAR::UTF32ToUTF8(kInvalidUTF32);`
Add more unittests from Google They were provided by Jeff Breidenbach <jbreiden@google.com>. Signed-off-by: Stefan Weil <sw@weilnetz.de> 2018-08-24 21:07:48 +08:00			`EXPECT_TRUE(utf8.empty());`
			`}`

			`} // namespace`