Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
6 changes: 3 additions & 3 deletions cpp/src/arrow/compute/kernels/scalar_cast_string.cc
Original file line number Diff line number Diff line change
Expand Up @@ -27,14 +27,14 @@
#include "arrow/util/formatting.h"
#include "arrow/util/int_util.h"
#include "arrow/util/optional.h"
#include "arrow/util/utf8.h"
#include "arrow/util/utf8_internal.h"
#include "arrow/visit_data_inline.h"

namespace arrow {

using internal::StringFormatter;
using util::InitializeUTF8;
using util::ValidateUTF8;
using util::ValidateUTF8Inline;

namespace compute {
namespace internal {
Expand Down Expand Up @@ -197,7 +197,7 @@ struct Utf8Validator {
Status VisitNull() { return Status::OK(); }

Status VisitValue(util::string_view str) {
if (ARROW_PREDICT_FALSE(!ValidateUTF8(str))) {
if (ARROW_PREDICT_FALSE(!ValidateUTF8Inline(str))) {
return Status::Invalid("Invalid UTF8 payload");
}
return Status::OK();
Expand Down
2 changes: 1 addition & 1 deletion cpp/src/arrow/compute/kernels/scalar_string_utf8.cc
Original file line number Diff line number Diff line change
Expand Up @@ -24,7 +24,7 @@
#endif

#include "arrow/compute/kernels/scalar_string_internal.h"
#include "arrow/util/utf8.h"
#include "arrow/util/utf8_internal.h"

namespace arrow {
namespace compute {
Expand Down
4 changes: 2 additions & 2 deletions cpp/src/arrow/csv/converter.cc
Original file line number Diff line number Diff line change
Expand Up @@ -37,7 +37,7 @@
#include "arrow/util/checked_cast.h"
#include "arrow/util/decimal.h"
#include "arrow/util/trie.h"
#include "arrow/util/utf8.h"
#include "arrow/util/utf8_internal.h"
#include "arrow/util/value_parsing.h" // IWYU pragma: keep

namespace arrow {
Expand Down Expand Up @@ -176,7 +176,7 @@ struct BinaryValueDecoder : public ValueDecoder {
}

Status Decode(const uint8_t* data, uint32_t size, bool quoted, value_type* out) {
if (CheckUTF8 && ARROW_PREDICT_FALSE(!util::ValidateUTF8(data, size))) {
if (CheckUTF8 && ARROW_PREDICT_FALSE(!util::ValidateUTF8Inline(data, size))) {
return Status::Invalid("CSV conversion error to ", type_->ToString(),
": invalid UTF8 data");
}
Expand Down
2 changes: 1 addition & 1 deletion cpp/src/arrow/csv/reader.cc
Original file line number Diff line number Diff line change
Expand Up @@ -49,7 +49,7 @@
#include "arrow/util/optional.h"
#include "arrow/util/task_group.h"
#include "arrow/util/thread_pool.h"
#include "arrow/util/utf8.h"
#include "arrow/util/utf8_internal.h"
#include "arrow/util/vector.h"

namespace arrow {
Expand Down
10 changes: 9 additions & 1 deletion cpp/src/arrow/util/utf8.cc
Original file line number Diff line number Diff line change
Expand Up @@ -15,6 +15,8 @@
// specific language governing permissions and limitations
// under the License.

#include "arrow/util/utf8.h"

#include <cstdint>
#include <iterator>
#include <mutex>
Expand All @@ -23,7 +25,7 @@

#include "arrow/result.h"
#include "arrow/util/logging.h"
#include "arrow/util/utf8.h"
#include "arrow/util/utf8_internal.h"
#include "arrow/vendored/utfcpp/checked.h"

// Can be defined by utfcpp
Expand Down Expand Up @@ -90,6 +92,12 @@ void InitializeUTF8() {
std::call_once(utf8_initialized, internal::InitializeLargeTable);
}

bool ValidateUTF8(const uint8_t* data, int64_t size) {
return ValidateUTF8Inline(data, size);
}

bool ValidateUTF8(const util::string_view& str) { return ValidateUTF8Inline(str); }

static const uint8_t kBOM[] = {0xEF, 0xBB, 0xBF};

Result<const uint8_t*> SkipUTF8BOM(const uint8_t* data, int64_t size) {
Expand Down
Loading