Skip to content
Draft
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
341 changes: 228 additions & 113 deletions python/cudf/cudf/tests/series/accessors/test_str.py
Original file line number Diff line number Diff line change
Expand Up @@ -2692,140 +2692,255 @@ def test_string_index_duplicate_str_cat(data, others, sep, na_rep, name):
)


def _assert_string_cat(data, others, sep, na_rep, index=None):
ps = pd.Series(data, index=index, dtype="str", name="nice name")
gs = cudf.Series(data, index=index, dtype="str", name="nice name")
is_any_others_ndarray = isinstance(others, (list, tuple)) and any(
isinstance(item, np.ndarray) for item in others
)

expect = ps.str.cat(others=others, sep=sep, na_rep=na_rep)
got = gs.str.cat(
others=_cat_convert_seq_to_cudf(others), sep=sep, na_rep=na_rep
)
if is_any_others_ndarray:
# pandas returns Index[object] which cuDF doesn't support
expect.index = expect.index.astype(got.index.dtype)
assert_eq(expect, got)


@pytest.mark.parametrize(
"others",
"data, sep, na_rep",
[
None,
["f", "g", "h", "i", "j"],
("f", "g", "h", "i", "j"),
pd.Series(["f", "g", "h", "i", "j"]),
pd.Series(["AbC", "de", "FGHI", "j", "kLm"]),
pd.Index(["f", "g", "h", "i", "j"]),
pd.Index(["AbC", "de", "FGHI", "j", "kLm"]),
(
np.array(["f", "g", "h", "i", "j"]),
np.array(["f", "g", "h", "i", "j"]),
pytest.param(["AbC", "de", "FGHI", "j", "kLm"], None, None),
pytest.param(["AbC", "de", "FGHI", "j", "kLm"], "", None),
pytest.param(["AbC", "de", "FGHI", "j", "kLm"], "|", None),
pytest.param(["AbC", "de", "FGHI", "j", "kLm"], "|||", None),
pytest.param(["nOPq", None, "RsT", None, "uVw"], "|", None),
pytest.param(["nOPq", None, "RsT", None, "uVw"], "|", ""),
pytest.param(["nOPq", None, "RsT", None, "uVw"], "|", "null"),
pytest.param([None, None, None, None, None], "|", "null"),
],
)
def test_string_cat_join(data, sep, na_rep):
_assert_string_cat(data, None, sep, na_rep)


@pytest.mark.parametrize(
"data, others, sep, na_rep",
[
pytest.param(
["nOPq", None, "RsT", None, "uVw"],
["f", "g", "h", "i", "j"],
None,
None,
),
[
np.array(["f", "g", "h", "i", "j"]),
np.array(["f", "g", "h", "i", "j"]),
],
[
pd.Series(["f", "g", "h", "i", "j"]),
pd.Series(["f", "g", "h", "i", "j"]),
],
(
pd.Series(["f", "g", "h", "i", "j"]),
pd.Series(["f", "g", "h", "i", "j"]),
pytest.param(
["nOPq", None, "RsT", None, "uVw"],
["f", "g", "h", "i", "j"],
"|",
"",
),
[
pd.Series(["f", "g", "h", "i", "j"]),
np.array(["f", "g", "h", "i", "j"]),
],
(
pytest.param(
["nOPq", None, "RsT", None, "uVw"],
pd.Series(["f", "g", "h", "i", "j"]),
np.array(["f", "g", "h", "i", "j"]),
"|",
"null",
),
(
pd.Series(["f", "g", "h", "i", "j"]),
np.array(["f", "a", "b", "f", "a"]),
pd.Series(["f", "g", "h", "i", "j"]),
np.array(["f", "a", "b", "f", "a"]),
np.array(["f", "a", "b", "f", "a"]),
pd.Index(["1", "2", "3", "4", "5"]),
np.array(["f", "a", "b", "f", "a"]),
pd.Index(["f", "g", "h", "i", "j"]),
pytest.param(
[None, None, None, None, None],
["f", "g", "h", "i", "j"],
"|",
None,
),
[
pd.Index(["f", "g", "h", "i", "j"]),
np.array(["f", "a", "b", "f", "a"]),
pd.Series(["f", "g", "h", "i", "j"]),
np.array(["f", "a", "b", "f", "a"]),
np.array(["f", "a", "b", "f", "a"]),
pd.Index(["f", "g", "h", "i", "j"]),
np.array(["f", "a", "b", "f", "a"]),
pytest.param(
[None, None, None, None, None],
pd.Index(["f", "g", "h", "i", "j"]),
],
[
pd.Series(["hello", "world", "abc", "xyz", "pqr"]),
pd.Series(["abc", "xyz", "hello", "pqr", "world"]),
],
[
pd.Series(
["hello", "world", "abc", "xyz", "pqr"],
index=[10, 11, 12, 13, 14],
),
pd.Series(
["abc", "xyz", "hello", "pqr", "world"],
index=[10, 15, 11, 13, 14],
"|",
"null",
),
],
)
def test_string_cat_elementwise_nulls(data, others, sep, na_rep):
_assert_string_cat(data, others, sep, na_rep)


@pytest.mark.parametrize(
"others",
[
pytest.param(["f", "g", "h", "i", "j"], id="list"),
pytest.param(("f", "g", "h", "i", "j"), id="tuple"),
pytest.param(pd.Series(["f", "g", "h", "i", "j"]), id="series"),
pytest.param(pd.Index(["f", "g", "h", "i", "j"]), id="index"),
pytest.param(
(
np.array(["f", "g", "h", "i", "j"]),
np.array(["f", "g", "h", "i", "j"]),
),
],
[
pd.Series(
["hello", "world", "abc", "xyz", "pqr"],
index=["10", "11", "12", "13", "14"],
id="tuple-of-ndarrays",
),
pytest.param(
[
np.array(["f", "g", "h", "i", "j"]),
np.array(["f", "g", "h", "i", "j"]),
],
id="list-of-ndarrays",
),
pytest.param(
[
pd.Series(["f", "g", "h", "i", "j"]),
pd.Series(["f", "g", "h", "i", "j"]),
],
id="list-of-series",
),
pytest.param(
(
pd.Series(["f", "g", "h", "i", "j"]),
pd.Series(["f", "g", "h", "i", "j"]),
),
pd.Series(
["abc", "xyz", "hello", "pqr", "world"],
index=["10", "11", "12", "13", "14"],
id="tuple-of-series",
),
pytest.param(
[
pd.Series(["f", "g", "h", "i", "j"]),
np.array(["f", "g", "h", "i", "j"]),
],
id="list-of-series-and-ndarray",
),
pytest.param(
(
pd.Series(["f", "g", "h", "i", "j"]),
np.array(["f", "g", "h", "i", "j"]),
),
],
[
pd.Series(
["hello", "world", "abc", "xyz", "pqr"],
index=["10", "11", "12", "13", "14"],
id="tuple-of-series-and-ndarray",
),
pytest.param(
(
pd.Series(["f", "g", "h", "i", "j"]),
np.array(["f", "a", "b", "f", "a"]),
pd.Series(["f", "g", "h", "i", "j"]),
np.array(["f", "a", "b", "f", "a"]),
np.array(["f", "a", "b", "f", "a"]),
pd.Index(["1", "2", "3", "4", "5"]),
np.array(["f", "a", "b", "f", "a"]),
pd.Index(["f", "g", "h", "i", "j"]),
),
pd.Series(
["abc", "xyz", "hello", "pqr", "world"],
index=["10", "15", "11", "13", "14"],
id="heterogeneous-tuple",
),
pytest.param(
[
pd.Index(["f", "g", "h", "i", "j"]),
np.array(["f", "a", "b", "f", "a"]),
pd.Series(["f", "g", "h", "i", "j"]),
np.array(["f", "a", "b", "f", "a"]),
np.array(["f", "a", "b", "f", "a"]),
pd.Index(["f", "g", "h", "i", "j"]),
np.array(["f", "a", "b", "f", "a"]),
pd.Index(["f", "g", "h", "i", "j"]),
],
id="heterogeneous-list",
),
],
)
def test_string_cat_input_forms(others):
_assert_string_cat(["AbC", "de", "FGHI", "j", "kLm"], others, "|", None)


@pytest.mark.parametrize(
"index, others",
[
pytest.param(
[10, 11, 12, 13, 14],
pd.Series(["f", "g", "h", "i", "j"]),
id="series-reindex",
),
pytest.param(
None,
[
pd.Series(["hello", "world", "abc", "xyz", "pqr"]),
pd.Series(["abc", "xyz", "hello", "pqr", "world"]),
],
id="multiple-series-default-index",
),
pytest.param(
None,
[
pd.Series(
["hello", "world", "abc", "xyz", "pqr"],
index=[10, 11, 12, 13, 14],
),
pd.Series(
["abc", "xyz", "hello", "pqr", "world"],
index=[10, 15, 11, 13, 14],
),
],
id="multiple-series-partial-integer-index",
),
pytest.param(
None,
[
pd.Series(
["hello", "world", "abc", "xyz", "pqr"],
index=["10", "11", "12", "13", "14"],
),
pd.Series(
["abc", "xyz", "hello", "pqr", "world"],
index=["10", "11", "12", "13", "14"],
),
],
marks=pytest.mark.xfail(
reason="https://github.com/NVIDIA/cudf/issues/21123"
),
],
[
pd.Series(
["hello", "world", "abc", "xyz", "pqr"],
index=["1", "2", "3", "4", "5"],
id="multiple-series-matching-string-index",
),
pytest.param(
None,
[
pd.Series(
["hello", "world", "abc", "xyz", "pqr"],
index=["10", "11", "12", "13", "14"],
),
pd.Series(
["abc", "xyz", "hello", "pqr", "world"],
index=["10", "15", "11", "13", "14"],
),
],
marks=pytest.mark.xfail(
reason="https://github.com/NVIDIA/cudf/issues/21123"
),
pd.Series(
["abc", "xyz", "hello", "pqr", "world"],
index=["10", "11", "12", "13", "14"],
id="multiple-series-partial-string-index",
),
pytest.param(
None,
[
pd.Series(
["hello", "world", "abc", "xyz", "pqr"],
index=["1", "2", "3", "4", "5"],
),
pd.Series(
["abc", "xyz", "hello", "pqr", "world"],
index=["10", "11", "12", "13", "14"],
),
],
marks=pytest.mark.xfail(
reason="https://github.com/NVIDIA/cudf/issues/21123"
),
],
id="multiple-series-disjoint-string-index",
),
],
)
@pytest.mark.parametrize("sep", [None, "", " ", "|", ",", "|||"])
@pytest.mark.parametrize("na_rep", [None, "", "null", "a"])
def test_string_cat(ps_gs, others, sep, na_rep, index, request):
# https://github.com/pandas-dev/pandas/issues/63371
ps, gs = ps_gs
is_any_others_ndarray = isinstance(others, (list, tuple)) and any(
isinstance(item, np.ndarray) for item in others
)
is_any_others_series_with_string_index = isinstance(
others, (list, tuple)
) and any(
isinstance(item, pd.Series)
and isinstance(item.index.dtype, pd.StringDtype)
for item in others
)
request.applymarker(
pytest.mark.xfail(
is_any_others_series_with_string_index,
reason="https://github.com/NVIDIA/cudf/issues/21123",
)
def test_string_cat_series_alignment(index, others):
_assert_string_cat(
["AbC", "de", "FGHI", "j", "kLm"], others, "|", None, index
)
pd_others = others
gd_others = _cat_convert_seq_to_cudf(others)

expect = ps.str.cat(others=pd_others, sep=sep, na_rep=na_rep)
got = gs.str.cat(others=gd_others, sep=sep, na_rep=na_rep)
if is_any_others_ndarray:
# pandas returns Index[object] which cuDF doesn't support
expect.index = expect.index.astype(got.index.dtype)
assert_eq(expect, got)

@pytest.mark.parametrize("sep", [None, "", " ", "|", ",", "|||"])
@pytest.mark.parametrize("na_rep", [None, "", "null", "a"])
def test_string_cat_index_others(data, sep, na_rep):
index = ["1", "2", "3", "4", "5"]
ps.index = index
gs.index = index
ps = pd.Series(data, index=index, dtype="str", name="nice name")
gs = cudf.Series(data, index=index, dtype="str", name="nice name")

expect = ps.str.cat(others=ps.index, sep=sep, na_rep=na_rep)
got = gs.str.cat(others=gs.index, sep=sep, na_rep=na_rep)
Expand Down
Loading