Skip to content
Closed
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
73 changes: 13 additions & 60 deletions cpp/src/arrow/compute/kernels/scalar_string.cc
Original file line number Diff line number Diff line change
Expand Up @@ -64,77 +64,30 @@ void StringDataTransform(KernelContext* ctx, const ExecBatch& batch,
}
}

// Generated with
//
// print("static constexpr uint8_t kAsciiUpperTable[] = {")
// for i in range(256):
// if i > 0: print(', ', end='')
// if i >= ord('a') and i <= ord('z'):
// print(i - 32, end='')
// else:
// print(i, end='')
// print("};")

static constexpr uint8_t kAsciiUpperTable[] = {
0, 1, 2, 3, 4, 5, 6, 7, 8, 9, 10, 11, 12, 13, 14, 15,
16, 17, 18, 19, 20, 21, 22, 23, 24, 25, 26, 27, 28, 29, 30, 31,
32, 33, 34, 35, 36, 37, 38, 39, 40, 41, 42, 43, 44, 45, 46, 47,
48, 49, 50, 51, 52, 53, 54, 55, 56, 57, 58, 59, 60, 61, 62, 63,
64, 65, 66, 67, 68, 69, 70, 71, 72, 73, 74, 75, 76, 77, 78, 79,
80, 81, 82, 83, 84, 85, 86, 87, 88, 89, 90, 91, 92, 93, 94, 95,
96, 65, 66, 67, 68, 69, 70, 71, 72, 73, 74, 75, 76, 77, 78, 79,
80, 81, 82, 83, 84, 85, 86, 87, 88, 89, 90, 123, 124, 125, 126, 127,
128, 129, 130, 131, 132, 133, 134, 135, 136, 137, 138, 139, 140, 141, 142, 143,
144, 145, 146, 147, 148, 149, 150, 151, 152, 153, 154, 155, 156, 157, 158, 159,
160, 161, 162, 163, 164, 165, 166, 167, 168, 169, 170, 171, 172, 173, 174, 175,
176, 177, 178, 179, 180, 181, 182, 183, 184, 185, 186, 187, 188, 189, 190, 191,
192, 193, 194, 195, 196, 197, 198, 199, 200, 201, 202, 203, 204, 205, 206, 207,
208, 209, 210, 211, 212, 213, 214, 215, 216, 217, 218, 219, 220, 221, 222, 223,
224, 225, 226, 227, 228, 229, 230, 231, 232, 233, 234, 235, 236, 237, 238, 239,
240, 241, 242, 243, 244, 245, 246, 247, 248, 249, 250, 251, 252, 253, 254, 255};

void TransformAsciiUpper(const uint8_t* input, int64_t length, uint8_t* output) {
for (int64_t i = 0; i < length; ++i) {
*output++ = kAsciiUpperTable[*input++];
const uint8_t utf8_code_unit = *input++;
// Code units in the range [a-z] can only be an encoding of an ascii
// character/codepoint, not the 2nd, 3rd or 4th code unit (byte) of an different
// codepoint. This guaranteed by non-overlap design of the unicode standard. (see
// section 2.5 of Unicode Standard Core Specification v13.0)
*output++ = ((utf8_code_unit >= 'a') && (utf8_code_unit <= 'z'))
? (utf8_code_unit - 32)
: utf8_code_unit;
}
}

void AsciiUpperExec(KernelContext* ctx, const ExecBatch& batch, Datum* out) {
StringDataTransform(ctx, batch, TransformAsciiUpper, out);
}

// Generated with
//
// print("static constexpr uint8_t kAsciiLowerTable[] = {")
// for i in range(256):
// if i > 0: print(', ', end='')
// if i >= ord('A') and i <= ord('Z'):
// print(i + 32, end='')
// else:
// print(i, end='')
// print("};")

static constexpr uint8_t kAsciiLowerTable[] = {
0, 1, 2, 3, 4, 5, 6, 7, 8, 9, 10, 11, 12, 13, 14, 15,
16, 17, 18, 19, 20, 21, 22, 23, 24, 25, 26, 27, 28, 29, 30, 31,
32, 33, 34, 35, 36, 37, 38, 39, 40, 41, 42, 43, 44, 45, 46, 47,
48, 49, 50, 51, 52, 53, 54, 55, 56, 57, 58, 59, 60, 61, 62, 63,
64, 97, 98, 99, 100, 101, 102, 103, 104, 105, 106, 107, 108, 109, 110, 111,
112, 113, 114, 115, 116, 117, 118, 119, 120, 121, 122, 91, 92, 93, 94, 95,
96, 97, 98, 99, 100, 101, 102, 103, 104, 105, 106, 107, 108, 109, 110, 111,
112, 113, 114, 115, 116, 117, 118, 119, 120, 121, 122, 123, 124, 125, 126, 127,
128, 129, 130, 131, 132, 133, 134, 135, 136, 137, 138, 139, 140, 141, 142, 143,
144, 145, 146, 147, 148, 149, 150, 151, 152, 153, 154, 155, 156, 157, 158, 159,
160, 161, 162, 163, 164, 165, 166, 167, 168, 169, 170, 171, 172, 173, 174, 175,
176, 177, 178, 179, 180, 181, 182, 183, 184, 185, 186, 187, 188, 189, 190, 191,
192, 193, 194, 195, 196, 197, 198, 199, 200, 201, 202, 203, 204, 205, 206, 207,
208, 209, 210, 211, 212, 213, 214, 215, 216, 217, 218, 219, 220, 221, 222, 223,
224, 225, 226, 227, 228, 229, 230, 231, 232, 233, 234, 235, 236, 237, 238, 239,
240, 241, 242, 243, 244, 245, 246, 247, 248, 249, 250, 251, 252, 253, 254, 255};

void TransformAsciiLower(const uint8_t* input, int64_t length, uint8_t* output) {
for (int64_t i = 0; i < length; ++i) {
*output++ = kAsciiLowerTable[*input++];
// As with TransformAsciiUpper, the same guarantee holds for the range [A-Z]
const uint8_t utf8_code_unit = *input++;
*output++ = ((utf8_code_unit >= 'A') && (utf8_code_unit <= 'Z'))
? (utf8_code_unit + 32)
: utf8_code_unit;
}
}

Expand Down
8 changes: 4 additions & 4 deletions cpp/src/arrow/compute/kernels/scalar_string_test.cc
Original file line number Diff line number Diff line change
Expand Up @@ -57,13 +57,13 @@ TYPED_TEST(TestStringKernels, AsciiLength) {
}

TYPED_TEST(TestStringKernels, AsciiUpper) {
this->CheckUnary("ascii_upper", "[\"aAa&\", null, \"\", \"b\"]", this->string_type(),
"[\"AAA&\", null, \"\", \"B\"]");
this->CheckUnary("ascii_upper", "[\"aAazZæÆ&\", null, \"\", \"b\"]",
this->string_type(), "[\"AAAZZæÆ&\", null, \"\", \"B\"]");
}

TYPED_TEST(TestStringKernels, AsciiLower) {
this->CheckUnary("ascii_lower", "[\"aAa&\", null, \"\", \"b\"]", this->string_type(),
"[\"aaa&\", null, \"\", \"b\"]");
this->CheckUnary("ascii_lower", "[\"aAazZæÆ&\", null, \"\", \"b\"]",
this->string_type(), "[\"aaazzæÆ&\", null, \"\", \"b\"]");
}

TYPED_TEST(TestStringKernels, Strptime) {
Expand Down