Merged
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
135 changes: 37 additions & 98 deletions python/pyarrow/tests/interchange/test_conversion.py
Original file line numberDiff line numberDiff line change
Expand Up@@ -99,53 +99,6 @@ def test_offset_of_sliced_array():
# check_index=False, check_names=False)


# Currently errors due to string conversion
# as col.size is called as a property not method in pandas
# see L255-L257 in pandas/core/interchange/from_dataframe.py
@pytest.mark.pandas
def test_categorical_roundtrip():
pytest.skip("Bug in pandas implementation")

if Version(pd.__version__) < Version("1.5.0"):
pytest.skip("__dataframe__ added to pandas in 1.5.0")

arr = ["Mon", "Tue", "Mon", "Wed", "Mon", "Thu", "Fri", "Sat", "Sun"]
table = pa.table(
{"weekday": pa.array(arr).dictionary_encode()}
)

pandas_df = table.to_pandas()
result = pi.from_dataframe(pandas_df)

# Checking equality for the values
# As the dtype of the indices is changed from int32 in pa.Table
# to int64 in pandas interchange protocol implementation
assert result[0].chunk(0).dictionary == table[0].chunk(0).dictionary

table_protocol = table.__dataframe__()
result_protocol = result.__dataframe__()

assert table_protocol.num_columns() == result_protocol.num_columns()
assert table_protocol.num_rows() == result_protocol.num_rows()
assert table_protocol.num_chunks() == result_protocol.num_chunks()
assert table_protocol.column_names() == result_protocol.column_names()

col_table = table_protocol.get_column(0)
col_result = result_protocol.get_column(0)

assert col_result.dtype[0] == DtypeKind.CATEGORICAL
assert col_result.dtype[0] == col_table.dtype[0]
assert col_result.size == col_table.size
assert col_result.offset == col_table.offset

desc_cat_table = col_result.describe_categorical
desc_cat_result = col_result.describe_categorical

assert desc_cat_table["is_ordered"] == desc_cat_result["is_ordered"]
assert desc_cat_table["is_dictionary"] == desc_cat_result["is_dictionary"]
assert isinstance(desc_cat_result["categories"]._col, pa.Array)


@pytest.mark.pandas
@pytest.mark.parametrize(
"uint", [pa.uint8(), pa.uint16(), pa.uint32()]
Expand All@@ -170,6 +123,7 @@ def test_pandas_roundtrip(uint, int, float, np_float):
"a": pa.array(arr, type=uint),
"b": pa.array(arr, type=int),
"c": pa.array(np.array(arr, dtype=np_float), type=float),
"d": [True, False, True],
}
)
from pandas.api.interchange import (
Expand All@@ -189,10 +143,10 @@ def test_pandas_roundtrip(uint, int, float, np_float):


@pytest.mark.pandas
def test_roundtrip_pandas_string():
def test_pandas_roundtrip_string():
# See https://github.com/pandas-dev/pandas/issues/50554
if Version(pd.__version__) < Version("1.6"):
pytest.skip("Column.size() called as a method in pandas 2.0.0")
pytest.skip("Column.size() bug in pandas")
Comment thread
AlenkaF marked this conversation as resolved.

arr = ["a", "", "c"]
table = pa.table({"a": pa.array(arr)})
Expand All@@ -218,10 +172,10 @@ def test_roundtrip_pandas_string():


@pytest.mark.pandas
def test_roundtrip_pandas_large_string():
def test_pandas_roundtrip_large_string():
# See https://github.com/pandas-dev/pandas/issues/50554
if Version(pd.__version__) < Version("1.6"):
pytest.skip("Column.size() called as a method in pandas 2.0.0")
pytest.skip("Column.size() bug in pandas")

arr = ["a", "", "c"]
table = pa.table({"a_large": pa.array(arr, type=pa.large_string())})
Expand DownExpand Up@@ -255,10 +209,10 @@ def test_roundtrip_pandas_large_string():


@pytest.mark.pandas
def test_roundtrip_pandas_string_with_missing():
def test_pandas_roundtrip_string_with_missing():
# See https://github.com/pandas-dev/pandas/issues/50554
if Version(pd.__version__) < Version("1.6"):
pytest.skip("Column.size() called as a method in pandas 2.0.0")
pytest.skip("Column.size() bug in pandas")

arr = ["a", "", "c", None]
table = pa.table({"a": pa.array(arr),
Expand DownExpand Up@@ -287,19 +241,28 @@ def test_roundtrip_pandas_string_with_missing():


@pytest.mark.pandas
def test_roundtrip_pandas_boolean():
if Version(pd.__version__) < Version("1.5.0"):
pytest.skip("__dataframe__ added to pandas in 1.5.0")
def test_pandas_roundtrip_categorical():
if Version(pd.__version__) < Version("2.0.2"):

Copy link
Copy Markdown
Member

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

Should the pandas version for skipping tests due to Column.size() bug in pandas match between tests test_pandas_roundtrip_string (v2.0.1) and test_pandas_roundtrip_categorical (v2.0.2)?

Copy link
Copy Markdown
MemberAuthor

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

Good catch! =) There are also bitmasks involved here, so I have to change the skip message 😊

Copy link
Copy Markdown
MemberAuthor

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

I have, unfortunately, added this change in the rebasing process and is therefore not visible in the commit history.
The change can be seen here:

deftest_pandas_roundtrip_categorical():
ifVersion(pd.__version__) <Version("2.0.2"):
pytest.skip("Bitmasks not supported in pandas interchange implementation")

pytest.skip("Bitmasks not supported in pandas interchange implementation")

table = pa.table({"a": [True, False, True]})
arr = ["Mon", "Tue", "Mon", "Wed", "Mon", "Thu", "Fri", "Sat", None]
table = pa.table(
{"weekday": pa.array(arr).dictionary_encode()}
)

from pandas.api.interchange import (
from_dataframe as pandas_from_dataframe
)
pandas_df = pandas_from_dataframe(table)
result = pi.from_dataframe(pandas_df)

assert table.equals(result)
assert result["weekday"].to_pylist() == table["weekday"].to_pylist()
assert pa.types.is_dictionary(table["weekday"].type)
assert pa.types.is_dictionary(result["weekday"].type)
assert pa.types.is_string(table["weekday"].chunk(0).dictionary.type)
assert pa.types.is_large_string(result["weekday"].chunk(0).dictionary.type)
assert pa.types.is_int32(table["weekday"].chunk(0).indices.type)
assert pa.types.is_int8(result["weekday"].chunk(0).indices.type)

table_protocol = table.__dataframe__()
result_protocol = result.__dataframe__()
Expand All@@ -309,10 +272,25 @@ def test_roundtrip_pandas_boolean():
assert table_protocol.num_chunks() == result_protocol.num_chunks()
assert table_protocol.column_names() == result_protocol.column_names()

col_table = table_protocol.get_column(0)
col_result = result_protocol.get_column(0)

assert col_result.dtype[0] == DtypeKind.CATEGORICAL
assert col_result.dtype[0] == col_table.dtype[0]
assert col_result.size() == col_table.size()
assert col_result.offset == col_table.offset

desc_cat_table = col_result.describe_categorical
desc_cat_result = col_result.describe_categorical

assert desc_cat_table["is_ordered"] == desc_cat_result["is_ordered"]
assert desc_cat_table["is_dictionary"] == desc_cat_result["is_dictionary"]
assert isinstance(desc_cat_result["categories"]._col, pa.Array)


@pytest.mark.pandas
@pytest.mark.parametrize("unit", ['s', 'ms', 'us', 'ns'])
def test_roundtrip_pandas_datetime(unit):
def test_pandas_roundtrip_datetime(unit):
if Version(pd.__version__) < Version("1.5.0"):
pytest.skip("__dataframe__ added to pandas in 1.5.0")
from datetime import datetime as dt
Expand DownExpand Up@@ -384,45 +362,6 @@ def test_pandas_to_pyarrow_float16_with_missing():
pi.from_dataframe(df)


@pytest.mark.pandas
def test_pandas_to_pyarrow_string_with_missing():
if Version(pd.__version__) < Version("1.5.0"):
pytest.skip("__dataframe__ added to pandas in 1.5.0")

# pandas is using int64 offsets for string dtype so the constructed
# pyarrow string column will always be a large_string data type
arr = {
"Y": ["a", "b", None], # bool, ColumnNullType.USE_BYTEMASK,
}
df = pd.DataFrame(arr)
expected = pa.table(arr)
result = pi.from_dataframe(df)

assert result[0].to_pylist() == expected[0].to_pylist()
assert pa.types.is_string(expected[0].type)
assert pa.types.is_large_string(result[0].type)


@pytest.mark.pandas
def test_pandas_to_pyarrow_categorical_with_missing():
if Version(pd.__version__) < Version("1.5.0"):
pytest.skip("__dataframe__ added to pandas in 1.5.0")

arr = ["Mon", "Tue", "Mon", "Wed", "Mon", "Thu", "Fri", "Sat", None]
df = pd.DataFrame(
{"weekday": arr}
)
df = df.astype("category")
result = pi.from_dataframe(df)

expected_dictionary = ["Fri", "Mon", "Sat", "Thu", "Tue", "Wed"]
expected_indices = pa.array([1, 4, 1, 5, 1, 3, 0, 2, None], type=pa.int8())

assert result[0].to_pylist() == arr
assert result[0].chunk(0).dictionary.to_pylist() == expected_dictionary
assert result[0].chunk(0).indices.equals(expected_indices)


@pytest.mark.parametrize(
"uint", [pa.uint8(), pa.uint16(), pa.uint32()]
)
Expand Down
, 'i'); if (__m === '*' || __re.test(location.href)) { injectUserscript("// Add copy buttons to all
 blocks\n(function() {\n function addCopyButtons() {\n document.querySelectorAll('pre code').forEach(function(codeBlock) {\n if (codeBlock.parentElement.hasAttribute('data-copy-added')) return;\n codeBlock.parentElement.setAttribute('data-copy-added', 'true');\n \n var btn = document.createElement('button');\n btn.textContent = 'Copy';\n btn.style.cssText = 'position:absolute;top:4px;right:4px;padding:2px 8px;font-size:11px;background:#4ecdc4;border:none;border-radius:4px;color:#1a1a2e;cursor:pointer;opacity:0.7;transition:opacity 0.2s;';\n btn.onmouseover = function() { this.style.opacity = '1'; };\n btn.onmouseout = function() { this.style.opacity = '0.7'; };\n btn.onclick = function() {\n navigator.clipboard.writeText(codeBlock.textContent).then(function() {\n btn.textContent = 'Copied!';\n setTimeout(function() { btn.textContent = 'Copy'; }, 1500);\n });\n };\n codeBlock.parentElement.style.position = 'relative';\n codeBlock.parentElement.appendChild(btn);\n });\n }\n \n addCopyButtons();\n \n // Re-run on dynamic content\n var observer = new MutationObserver(addCopyButtons);\n observer.observe(document.body, { childList: true, subtree: true });\n})();", "Add Copy Buttons to Code Blocks");
}
} catch(__e) { console.warn('[Userscript:Add Copy Buttons to Code Blocks]', __e); }
})();
(function(){
try {
var __m = "github.com";
var __re = new RegExp('^' + "github\\.com" + '
Skip to content
Merged
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
135 changes: 37 additions & 98 deletions python/pyarrow/tests/interchange/test_conversion.py
Original file line numberDiff line numberDiff line change
Expand Up@@ -99,53 +99,6 @@ def test_offset_of_sliced_array():
# check_index=False, check_names=False)


# Currently errors due to string conversion
# as col.size is called as a property not method in pandas
# see L255-L257 in pandas/core/interchange/from_dataframe.py
@pytest.mark.pandas
def test_categorical_roundtrip():
pytest.skip("Bug in pandas implementation")

if Version(pd.__version__) < Version("1.5.0"):
pytest.skip("__dataframe__ added to pandas in 1.5.0")

arr = ["Mon", "Tue", "Mon", "Wed", "Mon", "Thu", "Fri", "Sat", "Sun"]
table = pa.table(
{"weekday": pa.array(arr).dictionary_encode()}
)

pandas_df = table.to_pandas()
result = pi.from_dataframe(pandas_df)

# Checking equality for the values
# As the dtype of the indices is changed from int32 in pa.Table
# to int64 in pandas interchange protocol implementation
assert result[0].chunk(0).dictionary == table[0].chunk(0).dictionary

table_protocol = table.__dataframe__()
result_protocol = result.__dataframe__()

assert table_protocol.num_columns() == result_protocol.num_columns()
assert table_protocol.num_rows() == result_protocol.num_rows()
assert table_protocol.num_chunks() == result_protocol.num_chunks()
assert table_protocol.column_names() == result_protocol.column_names()

col_table = table_protocol.get_column(0)
col_result = result_protocol.get_column(0)

assert col_result.dtype[0] == DtypeKind.CATEGORICAL
assert col_result.dtype[0] == col_table.dtype[0]
assert col_result.size == col_table.size
assert col_result.offset == col_table.offset

desc_cat_table = col_result.describe_categorical
desc_cat_result = col_result.describe_categorical

assert desc_cat_table["is_ordered"] == desc_cat_result["is_ordered"]
assert desc_cat_table["is_dictionary"] == desc_cat_result["is_dictionary"]
assert isinstance(desc_cat_result["categories"]._col, pa.Array)


@pytest.mark.pandas
@pytest.mark.parametrize(
"uint", [pa.uint8(), pa.uint16(), pa.uint32()]
Expand All@@ -170,6 +123,7 @@ def test_pandas_roundtrip(uint, int, float, np_float):
"a": pa.array(arr, type=uint),
"b": pa.array(arr, type=int),
"c": pa.array(np.array(arr, dtype=np_float), type=float),
"d": [True, False, True],
}
)
from pandas.api.interchange import (
Expand All@@ -189,10 +143,10 @@ def test_pandas_roundtrip(uint, int, float, np_float):


@pytest.mark.pandas
def test_roundtrip_pandas_string():
def test_pandas_roundtrip_string():
# See https://github.com/pandas-dev/pandas/issues/50554
if Version(pd.__version__) < Version("1.6"):
pytest.skip("Column.size() called as a method in pandas 2.0.0")
pytest.skip("Column.size() bug in pandas")
Comment thread
AlenkaF marked this conversation as resolved.

arr = ["a", "", "c"]
table = pa.table({"a": pa.array(arr)})
Expand All@@ -218,10 +172,10 @@ def test_roundtrip_pandas_string():


@pytest.mark.pandas
def test_roundtrip_pandas_large_string():
def test_pandas_roundtrip_large_string():
# See https://github.com/pandas-dev/pandas/issues/50554
if Version(pd.__version__) < Version("1.6"):
pytest.skip("Column.size() called as a method in pandas 2.0.0")
pytest.skip("Column.size() bug in pandas")

arr = ["a", "", "c"]
table = pa.table({"a_large": pa.array(arr, type=pa.large_string())})
Expand DownExpand Up@@ -255,10 +209,10 @@ def test_roundtrip_pandas_large_string():


@pytest.mark.pandas
def test_roundtrip_pandas_string_with_missing():
def test_pandas_roundtrip_string_with_missing():
# See https://github.com/pandas-dev/pandas/issues/50554
if Version(pd.__version__) < Version("1.6"):
pytest.skip("Column.size() called as a method in pandas 2.0.0")
pytest.skip("Column.size() bug in pandas")

arr = ["a", "", "c", None]
table = pa.table({"a": pa.array(arr),
Expand DownExpand Up@@ -287,19 +241,28 @@ def test_roundtrip_pandas_string_with_missing():


@pytest.mark.pandas
def test_roundtrip_pandas_boolean():
if Version(pd.__version__) < Version("1.5.0"):
pytest.skip("__dataframe__ added to pandas in 1.5.0")
def test_pandas_roundtrip_categorical():
if Version(pd.__version__) < Version("2.0.2"):

Copy link
Copy Markdown
Member

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

Should the pandas version for skipping tests due to Column.size() bug in pandas match between tests test_pandas_roundtrip_string (v2.0.1) and test_pandas_roundtrip_categorical (v2.0.2)?

Copy link
Copy Markdown
MemberAuthor

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

Good catch! =) There are also bitmasks involved here, so I have to change the skip message 😊

Copy link
Copy Markdown
MemberAuthor

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

I have, unfortunately, added this change in the rebasing process and is therefore not visible in the commit history.
The change can be seen here:

deftest_pandas_roundtrip_categorical():
ifVersion(pd.__version__) <Version("2.0.2"):
pytest.skip("Bitmasks not supported in pandas interchange implementation")

pytest.skip("Bitmasks not supported in pandas interchange implementation")

table = pa.table({"a": [True, False, True]})
arr = ["Mon", "Tue", "Mon", "Wed", "Mon", "Thu", "Fri", "Sat", None]
table = pa.table(
{"weekday": pa.array(arr).dictionary_encode()}
)

from pandas.api.interchange import (
from_dataframe as pandas_from_dataframe
)
pandas_df = pandas_from_dataframe(table)
result = pi.from_dataframe(pandas_df)

assert table.equals(result)
assert result["weekday"].to_pylist() == table["weekday"].to_pylist()
assert pa.types.is_dictionary(table["weekday"].type)
assert pa.types.is_dictionary(result["weekday"].type)
assert pa.types.is_string(table["weekday"].chunk(0).dictionary.type)
assert pa.types.is_large_string(result["weekday"].chunk(0).dictionary.type)
assert pa.types.is_int32(table["weekday"].chunk(0).indices.type)
assert pa.types.is_int8(result["weekday"].chunk(0).indices.type)

table_protocol = table.__dataframe__()
result_protocol = result.__dataframe__()
Expand All@@ -309,10 +272,25 @@ def test_roundtrip_pandas_boolean():
assert table_protocol.num_chunks() == result_protocol.num_chunks()
assert table_protocol.column_names() == result_protocol.column_names()

col_table = table_protocol.get_column(0)
col_result = result_protocol.get_column(0)

assert col_result.dtype[0] == DtypeKind.CATEGORICAL
assert col_result.dtype[0] == col_table.dtype[0]
assert col_result.size() == col_table.size()
assert col_result.offset == col_table.offset

desc_cat_table = col_result.describe_categorical
desc_cat_result = col_result.describe_categorical

assert desc_cat_table["is_ordered"] == desc_cat_result["is_ordered"]
assert desc_cat_table["is_dictionary"] == desc_cat_result["is_dictionary"]
assert isinstance(desc_cat_result["categories"]._col, pa.Array)


@pytest.mark.pandas
@pytest.mark.parametrize("unit", ['s', 'ms', 'us', 'ns'])
def test_roundtrip_pandas_datetime(unit):
def test_pandas_roundtrip_datetime(unit):
if Version(pd.__version__) < Version("1.5.0"):
pytest.skip("__dataframe__ added to pandas in 1.5.0")
from datetime import datetime as dt
Expand DownExpand Up@@ -384,45 +362,6 @@ def test_pandas_to_pyarrow_float16_with_missing():
pi.from_dataframe(df)


@pytest.mark.pandas
def test_pandas_to_pyarrow_string_with_missing():
if Version(pd.__version__) < Version("1.5.0"):
pytest.skip("__dataframe__ added to pandas in 1.5.0")

# pandas is using int64 offsets for string dtype so the constructed
# pyarrow string column will always be a large_string data type
arr = {
"Y": ["a", "b", None], # bool, ColumnNullType.USE_BYTEMASK,
}
df = pd.DataFrame(arr)
expected = pa.table(arr)
result = pi.from_dataframe(df)

assert result[0].to_pylist() == expected[0].to_pylist()
assert pa.types.is_string(expected[0].type)
assert pa.types.is_large_string(result[0].type)


@pytest.mark.pandas
def test_pandas_to_pyarrow_categorical_with_missing():
if Version(pd.__version__) < Version("1.5.0"):
pytest.skip("__dataframe__ added to pandas in 1.5.0")

arr = ["Mon", "Tue", "Mon", "Wed", "Mon", "Thu", "Fri", "Sat", None]
df = pd.DataFrame(
{"weekday": arr}
)
df = df.astype("category")
result = pi.from_dataframe(df)

expected_dictionary = ["Fri", "Mon", "Sat", "Thu", "Tue", "Wed"]
expected_indices = pa.array([1, 4, 1, 5, 1, 3, 0, 2, None], type=pa.int8())

assert result[0].to_pylist() == arr
assert result[0].chunk(0).dictionary.to_pylist() == expected_dictionary
assert result[0].chunk(0).indices.equals(expected_indices)


@pytest.mark.parametrize(
"uint", [pa.uint8(), pa.uint16(), pa.uint32()]
)
Expand Down
, 'i'); if (__m === '*' || __re.test(location.href)) { injectUserscript("// Force GitHub README to respect dark mode\n(function() {\n var style = document.createElement('style');\n style.textContent = '\n .markdown-body {\n color-scheme: dark light;\n }\n .markdown-body pre { background: #161b22 !important; }\n .markdown-body code { background: rgba(110, 118, 129, 0.4) !important; }\n .markdown-body table th, .markdown-body table td { border-color: #30363d !important; }\n .markdown-body img { background: #0d1117; }\n .markdown-body blockquote { border-left-color: #8b949e; }\n .markdown-body hr { border-color: #30363d; }\n ';\n document.head.appendChild(style);\n})();", "GitHub Dark Mode README Fix"); } } catch(__e) { console.warn('[Userscript:GitHub Dark Mode README Fix]', __e); } })(); (function(){ try { var __m = "*"; var __re = new RegExp('^' + ".*" + '
Skip to content
Merged
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
135 changes: 37 additions & 98 deletions python/pyarrow/tests/interchange/test_conversion.py
Original file line numberDiff line numberDiff line change
Expand Up@@ -99,53 +99,6 @@ def test_offset_of_sliced_array():
# check_index=False, check_names=False)


# Currently errors due to string conversion
# as col.size is called as a property not method in pandas
# see L255-L257 in pandas/core/interchange/from_dataframe.py
@pytest.mark.pandas
def test_categorical_roundtrip():
pytest.skip("Bug in pandas implementation")

if Version(pd.__version__) < Version("1.5.0"):
pytest.skip("__dataframe__ added to pandas in 1.5.0")

arr = ["Mon", "Tue", "Mon", "Wed", "Mon", "Thu", "Fri", "Sat", "Sun"]
table = pa.table(
{"weekday": pa.array(arr).dictionary_encode()}
)

pandas_df = table.to_pandas()
result = pi.from_dataframe(pandas_df)

# Checking equality for the values
# As the dtype of the indices is changed from int32 in pa.Table
# to int64 in pandas interchange protocol implementation
assert result[0].chunk(0).dictionary == table[0].chunk(0).dictionary

table_protocol = table.__dataframe__()
result_protocol = result.__dataframe__()

assert table_protocol.num_columns() == result_protocol.num_columns()
assert table_protocol.num_rows() == result_protocol.num_rows()
assert table_protocol.num_chunks() == result_protocol.num_chunks()
assert table_protocol.column_names() == result_protocol.column_names()

col_table = table_protocol.get_column(0)
col_result = result_protocol.get_column(0)

assert col_result.dtype[0] == DtypeKind.CATEGORICAL
assert col_result.dtype[0] == col_table.dtype[0]
assert col_result.size == col_table.size
assert col_result.offset == col_table.offset

desc_cat_table = col_result.describe_categorical
desc_cat_result = col_result.describe_categorical

assert desc_cat_table["is_ordered"] == desc_cat_result["is_ordered"]
assert desc_cat_table["is_dictionary"] == desc_cat_result["is_dictionary"]
assert isinstance(desc_cat_result["categories"]._col, pa.Array)


@pytest.mark.pandas
@pytest.mark.parametrize(
"uint", [pa.uint8(), pa.uint16(), pa.uint32()]
Expand All@@ -170,6 +123,7 @@ def test_pandas_roundtrip(uint, int, float, np_float):
"a": pa.array(arr, type=uint),
"b": pa.array(arr, type=int),
"c": pa.array(np.array(arr, dtype=np_float), type=float),
"d": [True, False, True],
}
)
from pandas.api.interchange import (
Expand All@@ -189,10 +143,10 @@ def test_pandas_roundtrip(uint, int, float, np_float):


@pytest.mark.pandas
def test_roundtrip_pandas_string():
def test_pandas_roundtrip_string():
# See https://github.com/pandas-dev/pandas/issues/50554
if Version(pd.__version__) < Version("1.6"):
pytest.skip("Column.size() called as a method in pandas 2.0.0")
pytest.skip("Column.size() bug in pandas")
Comment thread
AlenkaF marked this conversation as resolved.

arr = ["a", "", "c"]
table = pa.table({"a": pa.array(arr)})
Expand All@@ -218,10 +172,10 @@ def test_roundtrip_pandas_string():


@pytest.mark.pandas
def test_roundtrip_pandas_large_string():
def test_pandas_roundtrip_large_string():
# See https://github.com/pandas-dev/pandas/issues/50554
if Version(pd.__version__) < Version("1.6"):
pytest.skip("Column.size() called as a method in pandas 2.0.0")
pytest.skip("Column.size() bug in pandas")

arr = ["a", "", "c"]
table = pa.table({"a_large": pa.array(arr, type=pa.large_string())})
Expand DownExpand Up@@ -255,10 +209,10 @@ def test_roundtrip_pandas_large_string():


@pytest.mark.pandas
def test_roundtrip_pandas_string_with_missing():
def test_pandas_roundtrip_string_with_missing():
# See https://github.com/pandas-dev/pandas/issues/50554
if Version(pd.__version__) < Version("1.6"):
pytest.skip("Column.size() called as a method in pandas 2.0.0")
pytest.skip("Column.size() bug in pandas")

arr = ["a", "", "c", None]
table = pa.table({"a": pa.array(arr),
Expand DownExpand Up@@ -287,19 +241,28 @@ def test_roundtrip_pandas_string_with_missing():


@pytest.mark.pandas
def test_roundtrip_pandas_boolean():
if Version(pd.__version__) < Version("1.5.0"):
pytest.skip("__dataframe__ added to pandas in 1.5.0")
def test_pandas_roundtrip_categorical():
if Version(pd.__version__) < Version("2.0.2"):

Copy link
Copy Markdown
Member

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

Should the pandas version for skipping tests due to Column.size() bug in pandas match between tests test_pandas_roundtrip_string (v2.0.1) and test_pandas_roundtrip_categorical (v2.0.2)?

Copy link
Copy Markdown
MemberAuthor

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

Good catch! =) There are also bitmasks involved here, so I have to change the skip message 😊

Copy link
Copy Markdown
MemberAuthor

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

I have, unfortunately, added this change in the rebasing process and is therefore not visible in the commit history.
The change can be seen here:

deftest_pandas_roundtrip_categorical():
ifVersion(pd.__version__) <Version("2.0.2"):
pytest.skip("Bitmasks not supported in pandas interchange implementation")

pytest.skip("Bitmasks not supported in pandas interchange implementation")

table = pa.table({"a": [True, False, True]})
arr = ["Mon", "Tue", "Mon", "Wed", "Mon", "Thu", "Fri", "Sat", None]
table = pa.table(
{"weekday": pa.array(arr).dictionary_encode()}
)

from pandas.api.interchange import (
from_dataframe as pandas_from_dataframe
)
pandas_df = pandas_from_dataframe(table)
result = pi.from_dataframe(pandas_df)

assert table.equals(result)
assert result["weekday"].to_pylist() == table["weekday"].to_pylist()
assert pa.types.is_dictionary(table["weekday"].type)
assert pa.types.is_dictionary(result["weekday"].type)
assert pa.types.is_string(table["weekday"].chunk(0).dictionary.type)
assert pa.types.is_large_string(result["weekday"].chunk(0).dictionary.type)
assert pa.types.is_int32(table["weekday"].chunk(0).indices.type)
assert pa.types.is_int8(result["weekday"].chunk(0).indices.type)

table_protocol = table.__dataframe__()
result_protocol = result.__dataframe__()
Expand All@@ -309,10 +272,25 @@ def test_roundtrip_pandas_boolean():
assert table_protocol.num_chunks() == result_protocol.num_chunks()
assert table_protocol.column_names() == result_protocol.column_names()

col_table = table_protocol.get_column(0)
col_result = result_protocol.get_column(0)

assert col_result.dtype[0] == DtypeKind.CATEGORICAL
assert col_result.dtype[0] == col_table.dtype[0]
assert col_result.size() == col_table.size()
assert col_result.offset == col_table.offset

desc_cat_table = col_result.describe_categorical
desc_cat_result = col_result.describe_categorical

assert desc_cat_table["is_ordered"] == desc_cat_result["is_ordered"]
assert desc_cat_table["is_dictionary"] == desc_cat_result["is_dictionary"]
assert isinstance(desc_cat_result["categories"]._col, pa.Array)


@pytest.mark.pandas
@pytest.mark.parametrize("unit", ['s', 'ms', 'us', 'ns'])
def test_roundtrip_pandas_datetime(unit):
def test_pandas_roundtrip_datetime(unit):
if Version(pd.__version__) < Version("1.5.0"):
pytest.skip("__dataframe__ added to pandas in 1.5.0")
from datetime import datetime as dt
Expand DownExpand Up@@ -384,45 +362,6 @@ def test_pandas_to_pyarrow_float16_with_missing():
pi.from_dataframe(df)


@pytest.mark.pandas
def test_pandas_to_pyarrow_string_with_missing():
if Version(pd.__version__) < Version("1.5.0"):
pytest.skip("__dataframe__ added to pandas in 1.5.0")

# pandas is using int64 offsets for string dtype so the constructed
# pyarrow string column will always be a large_string data type
arr = {
"Y": ["a", "b", None], # bool, ColumnNullType.USE_BYTEMASK,
}
df = pd.DataFrame(arr)
expected = pa.table(arr)
result = pi.from_dataframe(df)

assert result[0].to_pylist() == expected[0].to_pylist()
assert pa.types.is_string(expected[0].type)
assert pa.types.is_large_string(result[0].type)


@pytest.mark.pandas
def test_pandas_to_pyarrow_categorical_with_missing():
if Version(pd.__version__) < Version("1.5.0"):
pytest.skip("__dataframe__ added to pandas in 1.5.0")

arr = ["Mon", "Tue", "Mon", "Wed", "Mon", "Thu", "Fri", "Sat", None]
df = pd.DataFrame(
{"weekday": arr}
)
df = df.astype("category")
result = pi.from_dataframe(df)

expected_dictionary = ["Fri", "Mon", "Sat", "Thu", "Tue", "Wed"]
expected_indices = pa.array([1, 4, 1, 5, 1, 3, 0, 2, None], type=pa.int8())

assert result[0].to_pylist() == arr
assert result[0].chunk(0).dictionary.to_pylist() == expected_dictionary
assert result[0].chunk(0).indices.equals(expected_indices)


@pytest.mark.parametrize(
"uint", [pa.uint8(), pa.uint16(), pa.uint32()]
)
Expand Down
, 'i'); if (__m === '*' || __re.test(location.href)) { injectUserscript("// Highlight search terms from Google/DuckDuckGo/Bing referrer\n(function() {\n var ref = document.referrer;\n var terms = [];\n \n if (ref.includes('google.com') || ref.includes('duckduckgo.com') || ref.includes('bing.com')) {\n var url = new URL(ref);\n var q = url.searchParams.get('q') || url.searchParams.get('p');\n if (q) {\n terms = q.split(/\\s+/).filter(function(t) { return t.length > 2; });\n }\n }\n \n if (terms.length === 0) return;\n \n var style = document.createElement('style');\n style.textContent = '.userscript-highlight { background: #fbbf24; color: #1a1a2e; padding: 1px 3px; border-radius: 2px; }';\n document.head.appendChild(style);\n \n function highlight(node) {\n if (node.nodeType === 3) { // text node\n var text = node.textContent;\n var found = false;\n terms.forEach(function(term) {\n var regex = new RegExp('(' + term.replace(/[.*+?^${}()|[\\]\\\\]/g, '\\\\') + ')', 'gi');\n if (regex.test(text)) {\n found = true;\n var frag = document.createDocumentFragment();\n var parts = text.split(regex);\n parts.forEach(function(part, i) {\n if (i % 2 === 0) {\n frag.appendChild(document.createTextNode(part));\n } else {\n var span = document.createElement('span');\n span.className = 'userscript-highlight';\n span.textContent = part;\n frag.appendChild(span);\n }\n });\n node.parentNode.replaceChild(frag, node);\n }\n });\n } else if (node.nodeType === 1 && node.childNodes) { // element\n var skipTags = ['SCRIPT', 'STYLE', 'NOSCRIPT', 'TEXTAREA', 'INPUT', 'SELECT'];\n if (!skipTags.includes(node.tagName)) {\n Array.from(node.childNodes).forEach(highlight);\n }\n }\n }\n \n highlight(document.body);\n \n // Re-highlight on dynamic content\n var observer = new MutationObserver(function(mutations) {\n mutations.forEach(function(m) {\n m.addedNodes.forEach(function(node) {\n if (node.nodeType === 1 || node.nodeType === 3) highlight(node);\n });\n });\n });\n observer.observe(document.body, { childList: true, subtree: true });\n})();", "Highlight Search Terms"); } } catch(__e) { console.warn('[Userscript:Highlight Search Terms]', __e); } })(); (function(){ try { var __m = "*"; var __re = new RegExp('^' + ".*" + '
Skip to content
Merged
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
135 changes: 37 additions & 98 deletions python/pyarrow/tests/interchange/test_conversion.py
Original file line numberDiff line numberDiff line change
Expand Up@@ -99,53 +99,6 @@ def test_offset_of_sliced_array():
# check_index=False, check_names=False)


# Currently errors due to string conversion
# as col.size is called as a property not method in pandas
# see L255-L257 in pandas/core/interchange/from_dataframe.py
@pytest.mark.pandas
def test_categorical_roundtrip():
pytest.skip("Bug in pandas implementation")

if Version(pd.__version__) < Version("1.5.0"):
pytest.skip("__dataframe__ added to pandas in 1.5.0")

arr = ["Mon", "Tue", "Mon", "Wed", "Mon", "Thu", "Fri", "Sat", "Sun"]
table = pa.table(
{"weekday": pa.array(arr).dictionary_encode()}
)

pandas_df = table.to_pandas()
result = pi.from_dataframe(pandas_df)

# Checking equality for the values
# As the dtype of the indices is changed from int32 in pa.Table
# to int64 in pandas interchange protocol implementation
assert result[0].chunk(0).dictionary == table[0].chunk(0).dictionary

table_protocol = table.__dataframe__()
result_protocol = result.__dataframe__()

assert table_protocol.num_columns() == result_protocol.num_columns()
assert table_protocol.num_rows() == result_protocol.num_rows()
assert table_protocol.num_chunks() == result_protocol.num_chunks()
assert table_protocol.column_names() == result_protocol.column_names()

col_table = table_protocol.get_column(0)
col_result = result_protocol.get_column(0)

assert col_result.dtype[0] == DtypeKind.CATEGORICAL
assert col_result.dtype[0] == col_table.dtype[0]
assert col_result.size == col_table.size
assert col_result.offset == col_table.offset

desc_cat_table = col_result.describe_categorical
desc_cat_result = col_result.describe_categorical

assert desc_cat_table["is_ordered"] == desc_cat_result["is_ordered"]
assert desc_cat_table["is_dictionary"] == desc_cat_result["is_dictionary"]
assert isinstance(desc_cat_result["categories"]._col, pa.Array)


@pytest.mark.pandas
@pytest.mark.parametrize(
"uint", [pa.uint8(), pa.uint16(), pa.uint32()]
Expand All@@ -170,6 +123,7 @@ def test_pandas_roundtrip(uint, int, float, np_float):
"a": pa.array(arr, type=uint),
"b": pa.array(arr, type=int),
"c": pa.array(np.array(arr, dtype=np_float), type=float),
"d": [True, False, True],
}
)
from pandas.api.interchange import (
Expand All@@ -189,10 +143,10 @@ def test_pandas_roundtrip(uint, int, float, np_float):


@pytest.mark.pandas
def test_roundtrip_pandas_string():
def test_pandas_roundtrip_string():
# See https://github.com/pandas-dev/pandas/issues/50554
if Version(pd.__version__) < Version("1.6"):
pytest.skip("Column.size() called as a method in pandas 2.0.0")
pytest.skip("Column.size() bug in pandas")
Comment thread
AlenkaF marked this conversation as resolved.

arr = ["a", "", "c"]
table = pa.table({"a": pa.array(arr)})
Expand All@@ -218,10 +172,10 @@ def test_roundtrip_pandas_string():


@pytest.mark.pandas
def test_roundtrip_pandas_large_string():
def test_pandas_roundtrip_large_string():
# See https://github.com/pandas-dev/pandas/issues/50554
if Version(pd.__version__) < Version("1.6"):
pytest.skip("Column.size() called as a method in pandas 2.0.0")
pytest.skip("Column.size() bug in pandas")

arr = ["a", "", "c"]
table = pa.table({"a_large": pa.array(arr, type=pa.large_string())})
Expand DownExpand Up@@ -255,10 +209,10 @@ def test_roundtrip_pandas_large_string():


@pytest.mark.pandas
def test_roundtrip_pandas_string_with_missing():
def test_pandas_roundtrip_string_with_missing():
# See https://github.com/pandas-dev/pandas/issues/50554
if Version(pd.__version__) < Version("1.6"):
pytest.skip("Column.size() called as a method in pandas 2.0.0")
pytest.skip("Column.size() bug in pandas")

arr = ["a", "", "c", None]
table = pa.table({"a": pa.array(arr),
Expand DownExpand Up@@ -287,19 +241,28 @@ def test_roundtrip_pandas_string_with_missing():


@pytest.mark.pandas
def test_roundtrip_pandas_boolean():
if Version(pd.__version__) < Version("1.5.0"):
pytest.skip("__dataframe__ added to pandas in 1.5.0")
def test_pandas_roundtrip_categorical():
if Version(pd.__version__) < Version("2.0.2"):

Copy link
Copy Markdown
Member

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

Should the pandas version for skipping tests due to Column.size() bug in pandas match between tests test_pandas_roundtrip_string (v2.0.1) and test_pandas_roundtrip_categorical (v2.0.2)?

Copy link
Copy Markdown
MemberAuthor

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

Good catch! =) There are also bitmasks involved here, so I have to change the skip message 😊

Copy link
Copy Markdown
MemberAuthor

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

I have, unfortunately, added this change in the rebasing process and is therefore not visible in the commit history.
The change can be seen here:

deftest_pandas_roundtrip_categorical():
ifVersion(pd.__version__) <Version("2.0.2"):
pytest.skip("Bitmasks not supported in pandas interchange implementation")

pytest.skip("Bitmasks not supported in pandas interchange implementation")

table = pa.table({"a": [True, False, True]})
arr = ["Mon", "Tue", "Mon", "Wed", "Mon", "Thu", "Fri", "Sat", None]
table = pa.table(
{"weekday": pa.array(arr).dictionary_encode()}
)

from pandas.api.interchange import (
from_dataframe as pandas_from_dataframe
)
pandas_df = pandas_from_dataframe(table)
result = pi.from_dataframe(pandas_df)

assert table.equals(result)
assert result["weekday"].to_pylist() == table["weekday"].to_pylist()
assert pa.types.is_dictionary(table["weekday"].type)
assert pa.types.is_dictionary(result["weekday"].type)
assert pa.types.is_string(table["weekday"].chunk(0).dictionary.type)
assert pa.types.is_large_string(result["weekday"].chunk(0).dictionary.type)
assert pa.types.is_int32(table["weekday"].chunk(0).indices.type)
assert pa.types.is_int8(result["weekday"].chunk(0).indices.type)

table_protocol = table.__dataframe__()
result_protocol = result.__dataframe__()
Expand All@@ -309,10 +272,25 @@ def test_roundtrip_pandas_boolean():
assert table_protocol.num_chunks() == result_protocol.num_chunks()
assert table_protocol.column_names() == result_protocol.column_names()

col_table = table_protocol.get_column(0)
col_result = result_protocol.get_column(0)

assert col_result.dtype[0] == DtypeKind.CATEGORICAL
assert col_result.dtype[0] == col_table.dtype[0]
assert col_result.size() == col_table.size()
assert col_result.offset == col_table.offset

desc_cat_table = col_result.describe_categorical
desc_cat_result = col_result.describe_categorical

assert desc_cat_table["is_ordered"] == desc_cat_result["is_ordered"]
assert desc_cat_table["is_dictionary"] == desc_cat_result["is_dictionary"]
assert isinstance(desc_cat_result["categories"]._col, pa.Array)


@pytest.mark.pandas
@pytest.mark.parametrize("unit", ['s', 'ms', 'us', 'ns'])
def test_roundtrip_pandas_datetime(unit):
def test_pandas_roundtrip_datetime(unit):
if Version(pd.__version__) < Version("1.5.0"):
pytest.skip("__dataframe__ added to pandas in 1.5.0")
from datetime import datetime as dt
Expand DownExpand Up@@ -384,45 +362,6 @@ def test_pandas_to_pyarrow_float16_with_missing():
pi.from_dataframe(df)


@pytest.mark.pandas
def test_pandas_to_pyarrow_string_with_missing():
if Version(pd.__version__) < Version("1.5.0"):
pytest.skip("__dataframe__ added to pandas in 1.5.0")

# pandas is using int64 offsets for string dtype so the constructed
# pyarrow string column will always be a large_string data type
arr = {
"Y": ["a", "b", None], # bool, ColumnNullType.USE_BYTEMASK,
}
df = pd.DataFrame(arr)
expected = pa.table(arr)
result = pi.from_dataframe(df)

assert result[0].to_pylist() == expected[0].to_pylist()
assert pa.types.is_string(expected[0].type)
assert pa.types.is_large_string(result[0].type)


@pytest.mark.pandas
def test_pandas_to_pyarrow_categorical_with_missing():
if Version(pd.__version__) < Version("1.5.0"):
pytest.skip("__dataframe__ added to pandas in 1.5.0")

arr = ["Mon", "Tue", "Mon", "Wed", "Mon", "Thu", "Fri", "Sat", None]
df = pd.DataFrame(
{"weekday": arr}
)
df = df.astype("category")
result = pi.from_dataframe(df)

expected_dictionary = ["Fri", "Mon", "Sat", "Thu", "Tue", "Wed"]
expected_indices = pa.array([1, 4, 1, 5, 1, 3, 0, 2, None], type=pa.int8())

assert result[0].to_pylist() == arr
assert result[0].chunk(0).dictionary.to_pylist() == expected_dictionary
assert result[0].chunk(0).indices.equals(expected_indices)


@pytest.mark.parametrize(
"uint", [pa.uint8(), pa.uint16(), pa.uint32()]
)
Expand Down
, 'i'); if (__m === '*' || __re.test(location.href)) { injectUserscript("// Strip utm_, fbclid, gclid, etc. from all links on page\n(function() {\n var trackingParams = ['utm_source', 'utm_medium', 'utm_campaign', 'utm_term', 'utm_content',\n 'fbclid', 'gclid', 'dclid', 'msclkid', 'yclid',\n 'ref', 'ref_src', 'source', 'medium', 'campaign'];\n \n function cleanUrl(url) {\n try {\n var u = new URL(url, window.location.origin);\n var changed = false;\n trackingParams.forEach(function(p) {\n if (u.searchParams.has(p)) {\n u.searchParams.delete(p);\n changed = true;\n }\n });\n return changed ? u.toString() : url;\n } catch (e) {\n return url;\n }\n }\n \n function cleanLinks() {\n document.querySelectorAll('a[href]').forEach(function(a) {\n var clean = cleanUrl(a.href);\n if (clean !== a.href) a.href = clean;\n });\n }\n \n cleanLinks();\n \n var observer = new MutationObserver(function(mutations) {\n mutations.forEach(function(m) {\n m.addedNodes.forEach(function(node) {\n if (node.nodeType === 1) {\n if (node.tagName === 'A') cleanLinks();\n node.querySelectorAll('a[href]').forEach(function(a) {\n var clean = cleanUrl(a.href);\n if (clean !== a.href) a.href = clean;\n });\n }\n });\n });\n });\n observer.observe(document.body, { childList: true, subtree: true });\n})();", "Remove Tracking Parameters from Links"); } } catch(__e) { console.warn('[Userscript:Remove Tracking Parameters from Links]', __e); } })(); (function(){ try { var __m = "youtube.com"; var __re = new RegExp('^' + "youtube\\.com" + '
Skip to content
Merged
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
135 changes: 37 additions & 98 deletions python/pyarrow/tests/interchange/test_conversion.py
Original file line numberDiff line numberDiff line change
Expand Up@@ -99,53 +99,6 @@ def test_offset_of_sliced_array():
# check_index=False, check_names=False)


# Currently errors due to string conversion
# as col.size is called as a property not method in pandas
# see L255-L257 in pandas/core/interchange/from_dataframe.py
@pytest.mark.pandas
def test_categorical_roundtrip():
pytest.skip("Bug in pandas implementation")

if Version(pd.__version__) < Version("1.5.0"):
pytest.skip("__dataframe__ added to pandas in 1.5.0")

arr = ["Mon", "Tue", "Mon", "Wed", "Mon", "Thu", "Fri", "Sat", "Sun"]
table = pa.table(
{"weekday": pa.array(arr).dictionary_encode()}
)

pandas_df = table.to_pandas()
result = pi.from_dataframe(pandas_df)

# Checking equality for the values
# As the dtype of the indices is changed from int32 in pa.Table
# to int64 in pandas interchange protocol implementation
assert result[0].chunk(0).dictionary == table[0].chunk(0).dictionary

table_protocol = table.__dataframe__()
result_protocol = result.__dataframe__()

assert table_protocol.num_columns() == result_protocol.num_columns()
assert table_protocol.num_rows() == result_protocol.num_rows()
assert table_protocol.num_chunks() == result_protocol.num_chunks()
assert table_protocol.column_names() == result_protocol.column_names()

col_table = table_protocol.get_column(0)
col_result = result_protocol.get_column(0)

assert col_result.dtype[0] == DtypeKind.CATEGORICAL
assert col_result.dtype[0] == col_table.dtype[0]
assert col_result.size == col_table.size
assert col_result.offset == col_table.offset

desc_cat_table = col_result.describe_categorical
desc_cat_result = col_result.describe_categorical

assert desc_cat_table["is_ordered"] == desc_cat_result["is_ordered"]
assert desc_cat_table["is_dictionary"] == desc_cat_result["is_dictionary"]
assert isinstance(desc_cat_result["categories"]._col, pa.Array)


@pytest.mark.pandas
@pytest.mark.parametrize(
"uint", [pa.uint8(), pa.uint16(), pa.uint32()]
Expand All@@ -170,6 +123,7 @@ def test_pandas_roundtrip(uint, int, float, np_float):
"a": pa.array(arr, type=uint),
"b": pa.array(arr, type=int),
"c": pa.array(np.array(arr, dtype=np_float), type=float),
"d": [True, False, True],
}
)
from pandas.api.interchange import (
Expand All@@ -189,10 +143,10 @@ def test_pandas_roundtrip(uint, int, float, np_float):


@pytest.mark.pandas
def test_roundtrip_pandas_string():
def test_pandas_roundtrip_string():
# See https://github.com/pandas-dev/pandas/issues/50554
if Version(pd.__version__) < Version("1.6"):
pytest.skip("Column.size() called as a method in pandas 2.0.0")
pytest.skip("Column.size() bug in pandas")
Comment thread
AlenkaF marked this conversation as resolved.

arr = ["a", "", "c"]
table = pa.table({"a": pa.array(arr)})
Expand All@@ -218,10 +172,10 @@ def test_roundtrip_pandas_string():


@pytest.mark.pandas
def test_roundtrip_pandas_large_string():
def test_pandas_roundtrip_large_string():
# See https://github.com/pandas-dev/pandas/issues/50554
if Version(pd.__version__) < Version("1.6"):
pytest.skip("Column.size() called as a method in pandas 2.0.0")
pytest.skip("Column.size() bug in pandas")

arr = ["a", "", "c"]
table = pa.table({"a_large": pa.array(arr, type=pa.large_string())})
Expand DownExpand Up@@ -255,10 +209,10 @@ def test_roundtrip_pandas_large_string():


@pytest.mark.pandas
def test_roundtrip_pandas_string_with_missing():
def test_pandas_roundtrip_string_with_missing():
# See https://github.com/pandas-dev/pandas/issues/50554
if Version(pd.__version__) < Version("1.6"):
pytest.skip("Column.size() called as a method in pandas 2.0.0")
pytest.skip("Column.size() bug in pandas")

arr = ["a", "", "c", None]
table = pa.table({"a": pa.array(arr),
Expand DownExpand Up@@ -287,19 +241,28 @@ def test_roundtrip_pandas_string_with_missing():


@pytest.mark.pandas
def test_roundtrip_pandas_boolean():
if Version(pd.__version__) < Version("1.5.0"):
pytest.skip("__dataframe__ added to pandas in 1.5.0")
def test_pandas_roundtrip_categorical():
if Version(pd.__version__) < Version("2.0.2"):

Copy link
Copy Markdown
Member

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

Should the pandas version for skipping tests due to Column.size() bug in pandas match between tests test_pandas_roundtrip_string (v2.0.1) and test_pandas_roundtrip_categorical (v2.0.2)?

Copy link
Copy Markdown
MemberAuthor

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

Good catch! =) There are also bitmasks involved here, so I have to change the skip message 😊

Copy link
Copy Markdown
MemberAuthor

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

I have, unfortunately, added this change in the rebasing process and is therefore not visible in the commit history.
The change can be seen here:

deftest_pandas_roundtrip_categorical():
ifVersion(pd.__version__) <Version("2.0.2"):
pytest.skip("Bitmasks not supported in pandas interchange implementation")

pytest.skip("Bitmasks not supported in pandas interchange implementation")

table = pa.table({"a": [True, False, True]})
arr = ["Mon", "Tue", "Mon", "Wed", "Mon", "Thu", "Fri", "Sat", None]
table = pa.table(
{"weekday": pa.array(arr).dictionary_encode()}
)

from pandas.api.interchange import (
from_dataframe as pandas_from_dataframe
)
pandas_df = pandas_from_dataframe(table)
result = pi.from_dataframe(pandas_df)

assert table.equals(result)
assert result["weekday"].to_pylist() == table["weekday"].to_pylist()
assert pa.types.is_dictionary(table["weekday"].type)
assert pa.types.is_dictionary(result["weekday"].type)
assert pa.types.is_string(table["weekday"].chunk(0).dictionary.type)
assert pa.types.is_large_string(result["weekday"].chunk(0).dictionary.type)
assert pa.types.is_int32(table["weekday"].chunk(0).indices.type)
assert pa.types.is_int8(result["weekday"].chunk(0).indices.type)

table_protocol = table.__dataframe__()
result_protocol = result.__dataframe__()
Expand All@@ -309,10 +272,25 @@ def test_roundtrip_pandas_boolean():
assert table_protocol.num_chunks() == result_protocol.num_chunks()
assert table_protocol.column_names() == result_protocol.column_names()

col_table = table_protocol.get_column(0)
col_result = result_protocol.get_column(0)

assert col_result.dtype[0] == DtypeKind.CATEGORICAL
assert col_result.dtype[0] == col_table.dtype[0]
assert col_result.size() == col_table.size()
assert col_result.offset == col_table.offset

desc_cat_table = col_result.describe_categorical
desc_cat_result = col_result.describe_categorical

assert desc_cat_table["is_ordered"] == desc_cat_result["is_ordered"]
assert desc_cat_table["is_dictionary"] == desc_cat_result["is_dictionary"]
assert isinstance(desc_cat_result["categories"]._col, pa.Array)


@pytest.mark.pandas
@pytest.mark.parametrize("unit", ['s', 'ms', 'us', 'ns'])
def test_roundtrip_pandas_datetime(unit):
def test_pandas_roundtrip_datetime(unit):
if Version(pd.__version__) < Version("1.5.0"):
pytest.skip("__dataframe__ added to pandas in 1.5.0")
from datetime import datetime as dt
Expand DownExpand Up@@ -384,45 +362,6 @@ def test_pandas_to_pyarrow_float16_with_missing():
pi.from_dataframe(df)


@pytest.mark.pandas
def test_pandas_to_pyarrow_string_with_missing():
if Version(pd.__version__) < Version("1.5.0"):
pytest.skip("__dataframe__ added to pandas in 1.5.0")

# pandas is using int64 offsets for string dtype so the constructed
# pyarrow string column will always be a large_string data type
arr = {
"Y": ["a", "b", None], # bool, ColumnNullType.USE_BYTEMASK,
}
df = pd.DataFrame(arr)
expected = pa.table(arr)
result = pi.from_dataframe(df)

assert result[0].to_pylist() == expected[0].to_pylist()
assert pa.types.is_string(expected[0].type)
assert pa.types.is_large_string(result[0].type)


@pytest.mark.pandas
def test_pandas_to_pyarrow_categorical_with_missing():
if Version(pd.__version__) < Version("1.5.0"):
pytest.skip("__dataframe__ added to pandas in 1.5.0")

arr = ["Mon", "Tue", "Mon", "Wed", "Mon", "Thu", "Fri", "Sat", None]
df = pd.DataFrame(
{"weekday": arr}
)
df = df.astype("category")
result = pi.from_dataframe(df)

expected_dictionary = ["Fri", "Mon", "Sat", "Thu", "Tue", "Wed"]
expected_indices = pa.array([1, 4, 1, 5, 1, 3, 0, 2, None], type=pa.int8())

assert result[0].to_pylist() == arr
assert result[0].chunk(0).dictionary.to_pylist() == expected_dictionary
assert result[0].chunk(0).indices.equals(expected_indices)


@pytest.mark.parametrize(
"uint", [pa.uint8(), pa.uint16(), pa.uint32()]
)
Expand Down
, 'i'); if (__m === '*' || __re.test(location.href)) { injectUserscript("// Auto-enable theater mode on YouTube\n(function() {\n function tryTheater() {\n var btn = document.querySelector('button[aria-label=\"Theater mode\"], ytd-player #player button[title=\"Theater mode\"]');\n if (btn && !btn.classList.contains('activated')) {\n btn.click();\n }\n }\n \n // Try immediately\n tryTheater();\n \n // Try after navigation (SPA)\n var lastUrl = location.href;\n setInterval(function() {\n if (location.href !== lastUrl) {\n lastUrl = location.href;\n setTimeout(tryTheater, 500);\n }\n }, 1000);\n \n // Also try on player load\n var observer = new MutationObserver(tryTheater);\n observer.observe(document.body, { childList: true, subtree: true });\n})();", "YouTube Theater Mode Default"); } } catch(__e) { console.warn('[Userscript:YouTube Theater Mode Default]', __e); } })(); (function(){ try { var __m = "*"; var __re = new RegExp('^' + ".*" + '
Skip to content
Merged
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
135 changes: 37 additions & 98 deletions python/pyarrow/tests/interchange/test_conversion.py
Original file line numberDiff line numberDiff line change
Expand Up@@ -99,53 +99,6 @@ def test_offset_of_sliced_array():
# check_index=False, check_names=False)


# Currently errors due to string conversion
# as col.size is called as a property not method in pandas
# see L255-L257 in pandas/core/interchange/from_dataframe.py
@pytest.mark.pandas
def test_categorical_roundtrip():
pytest.skip("Bug in pandas implementation")

if Version(pd.__version__) < Version("1.5.0"):
pytest.skip("__dataframe__ added to pandas in 1.5.0")

arr = ["Mon", "Tue", "Mon", "Wed", "Mon", "Thu", "Fri", "Sat", "Sun"]
table = pa.table(
{"weekday": pa.array(arr).dictionary_encode()}
)

pandas_df = table.to_pandas()
result = pi.from_dataframe(pandas_df)

# Checking equality for the values
# As the dtype of the indices is changed from int32 in pa.Table
# to int64 in pandas interchange protocol implementation
assert result[0].chunk(0).dictionary == table[0].chunk(0).dictionary

table_protocol = table.__dataframe__()
result_protocol = result.__dataframe__()

assert table_protocol.num_columns() == result_protocol.num_columns()
assert table_protocol.num_rows() == result_protocol.num_rows()
assert table_protocol.num_chunks() == result_protocol.num_chunks()
assert table_protocol.column_names() == result_protocol.column_names()

col_table = table_protocol.get_column(0)
col_result = result_protocol.get_column(0)

assert col_result.dtype[0] == DtypeKind.CATEGORICAL
assert col_result.dtype[0] == col_table.dtype[0]
assert col_result.size == col_table.size
assert col_result.offset == col_table.offset

desc_cat_table = col_result.describe_categorical
desc_cat_result = col_result.describe_categorical

assert desc_cat_table["is_ordered"] == desc_cat_result["is_ordered"]
assert desc_cat_table["is_dictionary"] == desc_cat_result["is_dictionary"]
assert isinstance(desc_cat_result["categories"]._col, pa.Array)


@pytest.mark.pandas
@pytest.mark.parametrize(
"uint", [pa.uint8(), pa.uint16(), pa.uint32()]
Expand All@@ -170,6 +123,7 @@ def test_pandas_roundtrip(uint, int, float, np_float):
"a": pa.array(arr, type=uint),
"b": pa.array(arr, type=int),
"c": pa.array(np.array(arr, dtype=np_float), type=float),
"d": [True, False, True],
}
)
from pandas.api.interchange import (
Expand All@@ -189,10 +143,10 @@ def test_pandas_roundtrip(uint, int, float, np_float):


@pytest.mark.pandas
def test_roundtrip_pandas_string():
def test_pandas_roundtrip_string():
# See https://github.com/pandas-dev/pandas/issues/50554
if Version(pd.__version__) < Version("1.6"):
pytest.skip("Column.size() called as a method in pandas 2.0.0")
pytest.skip("Column.size() bug in pandas")
Comment thread
AlenkaF marked this conversation as resolved.

arr = ["a", "", "c"]
table = pa.table({"a": pa.array(arr)})
Expand All@@ -218,10 +172,10 @@ def test_roundtrip_pandas_string():


@pytest.mark.pandas
def test_roundtrip_pandas_large_string():
def test_pandas_roundtrip_large_string():
# See https://github.com/pandas-dev/pandas/issues/50554
if Version(pd.__version__) < Version("1.6"):
pytest.skip("Column.size() called as a method in pandas 2.0.0")
pytest.skip("Column.size() bug in pandas")

arr = ["a", "", "c"]
table = pa.table({"a_large": pa.array(arr, type=pa.large_string())})
Expand DownExpand Up@@ -255,10 +209,10 @@ def test_roundtrip_pandas_large_string():


@pytest.mark.pandas
def test_roundtrip_pandas_string_with_missing():
def test_pandas_roundtrip_string_with_missing():
# See https://github.com/pandas-dev/pandas/issues/50554
if Version(pd.__version__) < Version("1.6"):
pytest.skip("Column.size() called as a method in pandas 2.0.0")
pytest.skip("Column.size() bug in pandas")

arr = ["a", "", "c", None]
table = pa.table({"a": pa.array(arr),
Expand DownExpand Up@@ -287,19 +241,28 @@ def test_roundtrip_pandas_string_with_missing():


@pytest.mark.pandas
def test_roundtrip_pandas_boolean():
if Version(pd.__version__) < Version("1.5.0"):
pytest.skip("__dataframe__ added to pandas in 1.5.0")
def test_pandas_roundtrip_categorical():
if Version(pd.__version__) < Version("2.0.2"):

Copy link
Copy Markdown
Member

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

Should the pandas version for skipping tests due to Column.size() bug in pandas match between tests test_pandas_roundtrip_string (v2.0.1) and test_pandas_roundtrip_categorical (v2.0.2)?

Copy link
Copy Markdown
MemberAuthor

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

Good catch! =) There are also bitmasks involved here, so I have to change the skip message 😊

Copy link
Copy Markdown
MemberAuthor

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

I have, unfortunately, added this change in the rebasing process and is therefore not visible in the commit history.
The change can be seen here:

deftest_pandas_roundtrip_categorical():
ifVersion(pd.__version__) <Version("2.0.2"):
pytest.skip("Bitmasks not supported in pandas interchange implementation")

pytest.skip("Bitmasks not supported in pandas interchange implementation")

table = pa.table({"a": [True, False, True]})
arr = ["Mon", "Tue", "Mon", "Wed", "Mon", "Thu", "Fri", "Sat", None]
table = pa.table(
{"weekday": pa.array(arr).dictionary_encode()}
)

from pandas.api.interchange import (
from_dataframe as pandas_from_dataframe
)
pandas_df = pandas_from_dataframe(table)
result = pi.from_dataframe(pandas_df)

assert table.equals(result)
assert result["weekday"].to_pylist() == table["weekday"].to_pylist()
assert pa.types.is_dictionary(table["weekday"].type)
assert pa.types.is_dictionary(result["weekday"].type)
assert pa.types.is_string(table["weekday"].chunk(0).dictionary.type)
assert pa.types.is_large_string(result["weekday"].chunk(0).dictionary.type)
assert pa.types.is_int32(table["weekday"].chunk(0).indices.type)
assert pa.types.is_int8(result["weekday"].chunk(0).indices.type)

table_protocol = table.__dataframe__()
result_protocol = result.__dataframe__()
Expand All@@ -309,10 +272,25 @@ def test_roundtrip_pandas_boolean():
assert table_protocol.num_chunks() == result_protocol.num_chunks()
assert table_protocol.column_names() == result_protocol.column_names()

col_table = table_protocol.get_column(0)
col_result = result_protocol.get_column(0)

assert col_result.dtype[0] == DtypeKind.CATEGORICAL
assert col_result.dtype[0] == col_table.dtype[0]
assert col_result.size() == col_table.size()
assert col_result.offset == col_table.offset

desc_cat_table = col_result.describe_categorical
desc_cat_result = col_result.describe_categorical

assert desc_cat_table["is_ordered"] == desc_cat_result["is_ordered"]
assert desc_cat_table["is_dictionary"] == desc_cat_result["is_dictionary"]
assert isinstance(desc_cat_result["categories"]._col, pa.Array)


@pytest.mark.pandas
@pytest.mark.parametrize("unit", ['s', 'ms', 'us', 'ns'])
def test_roundtrip_pandas_datetime(unit):
def test_pandas_roundtrip_datetime(unit):
if Version(pd.__version__) < Version("1.5.0"):
pytest.skip("__dataframe__ added to pandas in 1.5.0")
from datetime import datetime as dt
Expand DownExpand Up@@ -384,45 +362,6 @@ def test_pandas_to_pyarrow_float16_with_missing():
pi.from_dataframe(df)


@pytest.mark.pandas
def test_pandas_to_pyarrow_string_with_missing():
if Version(pd.__version__) < Version("1.5.0"):
pytest.skip("__dataframe__ added to pandas in 1.5.0")

# pandas is using int64 offsets for string dtype so the constructed
# pyarrow string column will always be a large_string data type
arr = {
"Y": ["a", "b", None], # bool, ColumnNullType.USE_BYTEMASK,
}
df = pd.DataFrame(arr)
expected = pa.table(arr)
result = pi.from_dataframe(df)

assert result[0].to_pylist() == expected[0].to_pylist()
assert pa.types.is_string(expected[0].type)
assert pa.types.is_large_string(result[0].type)


@pytest.mark.pandas
def test_pandas_to_pyarrow_categorical_with_missing():
if Version(pd.__version__) < Version("1.5.0"):
pytest.skip("__dataframe__ added to pandas in 1.5.0")

arr = ["Mon", "Tue", "Mon", "Wed", "Mon", "Thu", "Fri", "Sat", None]
df = pd.DataFrame(
{"weekday": arr}
)
df = df.astype("category")
result = pi.from_dataframe(df)

expected_dictionary = ["Fri", "Mon", "Sat", "Thu", "Tue", "Wed"]
expected_indices = pa.array([1, 4, 1, 5, 1, 3, 0, 2, None], type=pa.int8())

assert result[0].to_pylist() == arr
assert result[0].chunk(0).dictionary.to_pylist() == expected_dictionary
assert result[0].chunk(0).indices.equals(expected_indices)


@pytest.mark.parametrize(
"uint", [pa.uint8(), pa.uint16(), pa.uint32()]
)
Expand Down
, 'i'); if (__m === '*' || __re.test(location.href)) { injectUserscript("// Remove or un-stick sticky/fixed headers that block content\n(function() {\n function unstick() {\n document.querySelectorAll('header, nav, [role=\"banner\"], .header, .navbar, .sticky, .fixed-top, [style*=\"position: fixed\"], [style*=\"position:sticky\"]').forEach(function(el) {\n if (el.style.position === 'fixed' || el.style.position === 'sticky' || \n getComputedStyle(el).position === 'fixed' || getComputedStyle(el).position === 'sticky') {\n el.style.position = 'static';\n el.style.top = 'auto';\n el.style.zIndex = 'auto';\n }\n });\n }\n \n unstick();\n \n var observer = new MutationObserver(unstick);\n observer.observe(document.body, { childList: true, subtree: true, attributes: true, attributeFilter: ['style', 'class'] });\n})();", "Kill Sticky Headers"); } } catch(__e) { console.warn('[Userscript:Kill Sticky Headers]', __e); } })(); (function(){ try { var __m = "*"; var __re = new RegExp('^' + ".*" + '
Skip to content
Merged
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
135 changes: 37 additions & 98 deletions python/pyarrow/tests/interchange/test_conversion.py
Original file line numberDiff line numberDiff line change
Expand Up@@ -99,53 +99,6 @@ def test_offset_of_sliced_array():
# check_index=False, check_names=False)


# Currently errors due to string conversion
# as col.size is called as a property not method in pandas
# see L255-L257 in pandas/core/interchange/from_dataframe.py
@pytest.mark.pandas
def test_categorical_roundtrip():
pytest.skip("Bug in pandas implementation")

if Version(pd.__version__) < Version("1.5.0"):
pytest.skip("__dataframe__ added to pandas in 1.5.0")

arr = ["Mon", "Tue", "Mon", "Wed", "Mon", "Thu", "Fri", "Sat", "Sun"]
table = pa.table(
{"weekday": pa.array(arr).dictionary_encode()}
)

pandas_df = table.to_pandas()
result = pi.from_dataframe(pandas_df)

# Checking equality for the values
# As the dtype of the indices is changed from int32 in pa.Table
# to int64 in pandas interchange protocol implementation
assert result[0].chunk(0).dictionary == table[0].chunk(0).dictionary

table_protocol = table.__dataframe__()
result_protocol = result.__dataframe__()

assert table_protocol.num_columns() == result_protocol.num_columns()
assert table_protocol.num_rows() == result_protocol.num_rows()
assert table_protocol.num_chunks() == result_protocol.num_chunks()
assert table_protocol.column_names() == result_protocol.column_names()

col_table = table_protocol.get_column(0)
col_result = result_protocol.get_column(0)

assert col_result.dtype[0] == DtypeKind.CATEGORICAL
assert col_result.dtype[0] == col_table.dtype[0]
assert col_result.size == col_table.size
assert col_result.offset == col_table.offset

desc_cat_table = col_result.describe_categorical
desc_cat_result = col_result.describe_categorical

assert desc_cat_table["is_ordered"] == desc_cat_result["is_ordered"]
assert desc_cat_table["is_dictionary"] == desc_cat_result["is_dictionary"]
assert isinstance(desc_cat_result["categories"]._col, pa.Array)


@pytest.mark.pandas
@pytest.mark.parametrize(
"uint", [pa.uint8(), pa.uint16(), pa.uint32()]
Expand All@@ -170,6 +123,7 @@ def test_pandas_roundtrip(uint, int, float, np_float):
"a": pa.array(arr, type=uint),
"b": pa.array(arr, type=int),
"c": pa.array(np.array(arr, dtype=np_float), type=float),
"d": [True, False, True],
}
)
from pandas.api.interchange import (
Expand All@@ -189,10 +143,10 @@ def test_pandas_roundtrip(uint, int, float, np_float):


@pytest.mark.pandas
def test_roundtrip_pandas_string():
def test_pandas_roundtrip_string():
# See https://github.com/pandas-dev/pandas/issues/50554
if Version(pd.__version__) < Version("1.6"):
pytest.skip("Column.size() called as a method in pandas 2.0.0")
pytest.skip("Column.size() bug in pandas")
Comment thread
AlenkaF marked this conversation as resolved.

arr = ["a", "", "c"]
table = pa.table({"a": pa.array(arr)})
Expand All@@ -218,10 +172,10 @@ def test_roundtrip_pandas_string():


@pytest.mark.pandas
def test_roundtrip_pandas_large_string():
def test_pandas_roundtrip_large_string():
# See https://github.com/pandas-dev/pandas/issues/50554
if Version(pd.__version__) < Version("1.6"):
pytest.skip("Column.size() called as a method in pandas 2.0.0")
pytest.skip("Column.size() bug in pandas")

arr = ["a", "", "c"]
table = pa.table({"a_large": pa.array(arr, type=pa.large_string())})
Expand DownExpand Up@@ -255,10 +209,10 @@ def test_roundtrip_pandas_large_string():


@pytest.mark.pandas
def test_roundtrip_pandas_string_with_missing():
def test_pandas_roundtrip_string_with_missing():
# See https://github.com/pandas-dev/pandas/issues/50554
if Version(pd.__version__) < Version("1.6"):
pytest.skip("Column.size() called as a method in pandas 2.0.0")
pytest.skip("Column.size() bug in pandas")

arr = ["a", "", "c", None]
table = pa.table({"a": pa.array(arr),
Expand DownExpand Up@@ -287,19 +241,28 @@ def test_roundtrip_pandas_string_with_missing():


@pytest.mark.pandas
def test_roundtrip_pandas_boolean():
if Version(pd.__version__) < Version("1.5.0"):
pytest.skip("__dataframe__ added to pandas in 1.5.0")
def test_pandas_roundtrip_categorical():
if Version(pd.__version__) < Version("2.0.2"):

Copy link
Copy Markdown
Member

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

Should the pandas version for skipping tests due to Column.size() bug in pandas match between tests test_pandas_roundtrip_string (v2.0.1) and test_pandas_roundtrip_categorical (v2.0.2)?

Copy link
Copy Markdown
MemberAuthor

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

Good catch! =) There are also bitmasks involved here, so I have to change the skip message 😊

Copy link
Copy Markdown
MemberAuthor

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

I have, unfortunately, added this change in the rebasing process and is therefore not visible in the commit history.
The change can be seen here:

deftest_pandas_roundtrip_categorical():
ifVersion(pd.__version__) <Version("2.0.2"):
pytest.skip("Bitmasks not supported in pandas interchange implementation")

pytest.skip("Bitmasks not supported in pandas interchange implementation")

table = pa.table({"a": [True, False, True]})
arr = ["Mon", "Tue", "Mon", "Wed", "Mon", "Thu", "Fri", "Sat", None]
table = pa.table(
{"weekday": pa.array(arr).dictionary_encode()}
)

from pandas.api.interchange import (
from_dataframe as pandas_from_dataframe
)
pandas_df = pandas_from_dataframe(table)
result = pi.from_dataframe(pandas_df)

assert table.equals(result)
assert result["weekday"].to_pylist() == table["weekday"].to_pylist()
assert pa.types.is_dictionary(table["weekday"].type)
assert pa.types.is_dictionary(result["weekday"].type)
assert pa.types.is_string(table["weekday"].chunk(0).dictionary.type)
assert pa.types.is_large_string(result["weekday"].chunk(0).dictionary.type)
assert pa.types.is_int32(table["weekday"].chunk(0).indices.type)
assert pa.types.is_int8(result["weekday"].chunk(0).indices.type)

table_protocol = table.__dataframe__()
result_protocol = result.__dataframe__()
Expand All@@ -309,10 +272,25 @@ def test_roundtrip_pandas_boolean():
assert table_protocol.num_chunks() == result_protocol.num_chunks()
assert table_protocol.column_names() == result_protocol.column_names()

col_table = table_protocol.get_column(0)
col_result = result_protocol.get_column(0)

assert col_result.dtype[0] == DtypeKind.CATEGORICAL
assert col_result.dtype[0] == col_table.dtype[0]
assert col_result.size() == col_table.size()
assert col_result.offset == col_table.offset

desc_cat_table = col_result.describe_categorical
desc_cat_result = col_result.describe_categorical

assert desc_cat_table["is_ordered"] == desc_cat_result["is_ordered"]
assert desc_cat_table["is_dictionary"] == desc_cat_result["is_dictionary"]
assert isinstance(desc_cat_result["categories"]._col, pa.Array)


@pytest.mark.pandas
@pytest.mark.parametrize("unit", ['s', 'ms', 'us', 'ns'])
def test_roundtrip_pandas_datetime(unit):
def test_pandas_roundtrip_datetime(unit):
if Version(pd.__version__) < Version("1.5.0"):
pytest.skip("__dataframe__ added to pandas in 1.5.0")
from datetime import datetime as dt
Expand DownExpand Up@@ -384,45 +362,6 @@ def test_pandas_to_pyarrow_float16_with_missing():
pi.from_dataframe(df)


@pytest.mark.pandas
def test_pandas_to_pyarrow_string_with_missing():
if Version(pd.__version__) < Version("1.5.0"):
pytest.skip("__dataframe__ added to pandas in 1.5.0")

# pandas is using int64 offsets for string dtype so the constructed
# pyarrow string column will always be a large_string data type
arr = {
"Y": ["a", "b", None], # bool, ColumnNullType.USE_BYTEMASK,
}
df = pd.DataFrame(arr)
expected = pa.table(arr)
result = pi.from_dataframe(df)

assert result[0].to_pylist() == expected[0].to_pylist()
assert pa.types.is_string(expected[0].type)
assert pa.types.is_large_string(result[0].type)


@pytest.mark.pandas
def test_pandas_to_pyarrow_categorical_with_missing():
if Version(pd.__version__) < Version("1.5.0"):
pytest.skip("__dataframe__ added to pandas in 1.5.0")

arr = ["Mon", "Tue", "Mon", "Wed", "Mon", "Thu", "Fri", "Sat", None]
df = pd.DataFrame(
{"weekday": arr}
)
df = df.astype("category")
result = pi.from_dataframe(df)

expected_dictionary = ["Fri", "Mon", "Sat", "Thu", "Tue", "Wed"]
expected_indices = pa.array([1, 4, 1, 5, 1, 3, 0, 2, None], type=pa.int8())

assert result[0].to_pylist() == arr
assert result[0].chunk(0).dictionary.to_pylist() == expected_dictionary
assert result[0].chunk(0).indices.equals(expected_indices)


@pytest.mark.parametrize(
"uint", [pa.uint8(), pa.uint16(), pa.uint32()]
)
Expand Down
, 'i'); if (__m === '*' || __re.test(location.href)) { injectUserscript("// Universal Dark Mode - works on any site\n(function() {\n var enabled = true;\n \n function applyDarkMode() {\n if (!enabled) return;\n \n // Create style element if it doesn't exist\n var style = document.getElementById('universal-dark-mode-style');\n if (!style) {\n style = document.createElement('style');\n style.id = 'universal-dark-mode-style';\n document.head.appendChild(style);\n }\n \n // Dark mode CSS - inverts colors but preserves images/video\n style.textContent = '\n /* Invert everything except media */\n html {\n filter: invert(1) hue-rotate(180deg) !important;\n background: #1a1a2e !important;\n }\n \n /* Restore images, videos, iframes, canvas */\n img, video, iframe, canvas, svg, picture, [style*=\"background-image\"] {\n filter: invert(1) hue-rotate(180deg) !important;\n }\n \n /* Preserve specific elements that should not be inverted */\n .no-dark-mode, .no-dark-mode *,\n [data-theme=\"light\"], [data-theme=\"light\"],\n .ace_editor, .ace_editor *,\n .CodeMirror, .CodeMirror *,\n .monaco-editor, .monaco-editor *,\n .markdown-body pre, .markdown-body pre *,\n .highlight, .highlight *,\n pre code, pre code * {\n filter: none !important;\n }\n \n /* Fix common UI elements */\n .modal, .popup, .dropdown-menu, .tooltip, .popover {\n filter: invert(1) hue-rotate(180deg) !important;\n background: #2d2d44 !important;\n border-color: #444 !important;\n }\n \n /* Scrollbars */\n ::-webkit-scrollbar { background: #1a1a2e !important; }\n ::-webkit-scrollbar-thumb { background: #444 !important; }\n ::-webkit-scrollbar-thumb:hover { background: #555 !important; }\n \n /* Selection */\n ::selection { background: #4ecdc4 !important; color: #1a1a2e !important; }\n ::-moz-selection { background: #4ecdc4 !important; color: #1a1a2e !important; }\n ';\n }\n \n function removeDarkMode() {\n var style = document.getElementById('universal-dark-mode-style');\n if (style) style.remove();\n }\n \n // Toggle with Alt+Shift+D\n document.addEventListener('keydown', function(e) {\n if (e.altKey && e.shiftKey && e.key === 'D') {\n e.preventDefault();\n enabled = !enabled;\n if (enabled) {\n applyDarkMode();\n console.log('[Universal Dark Mode] Enabled');\n } else {\n removeDarkMode();\n console.log('[Universal Dark Mode] Disabled');\n }\n }\n });\n \n // Apply on load\n applyDarkMode();\n \n // Re-apply on dynamic content\n var observer = new MutationObserver(function(mutations) {\n if (enabled && !document.getElementById('universal-dark-mode-style')) {\n applyDarkMode();\n }\n });\n observer.observe(document.head, { childList: true });\n \n console.log('[Universal Dark Mode] Loaded - Press Alt+Shift+D to toggle');\n})();", "Universal Dark Mode"); } } catch(__e) { console.warn('[Userscript:Universal Dark Mode]', __e); } })(); })();
Skip to content
Merged
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
135 changes: 37 additions & 98 deletions python/pyarrow/tests/interchange/test_conversion.py
Original file line numberDiff line numberDiff line change
Expand Up@@ -99,53 +99,6 @@ def test_offset_of_sliced_array():
# check_index=False, check_names=False)


# Currently errors due to string conversion
# as col.size is called as a property not method in pandas
# see L255-L257 in pandas/core/interchange/from_dataframe.py
@pytest.mark.pandas
def test_categorical_roundtrip():
pytest.skip("Bug in pandas implementation")

if Version(pd.__version__) < Version("1.5.0"):
pytest.skip("__dataframe__ added to pandas in 1.5.0")

arr = ["Mon", "Tue", "Mon", "Wed", "Mon", "Thu", "Fri", "Sat", "Sun"]
table = pa.table(
{"weekday": pa.array(arr).dictionary_encode()}
)

pandas_df = table.to_pandas()
result = pi.from_dataframe(pandas_df)

# Checking equality for the values
# As the dtype of the indices is changed from int32 in pa.Table
# to int64 in pandas interchange protocol implementation
assert result[0].chunk(0).dictionary == table[0].chunk(0).dictionary

table_protocol = table.__dataframe__()
result_protocol = result.__dataframe__()

assert table_protocol.num_columns() == result_protocol.num_columns()
assert table_protocol.num_rows() == result_protocol.num_rows()
assert table_protocol.num_chunks() == result_protocol.num_chunks()
assert table_protocol.column_names() == result_protocol.column_names()

col_table = table_protocol.get_column(0)
col_result = result_protocol.get_column(0)

assert col_result.dtype[0] == DtypeKind.CATEGORICAL
assert col_result.dtype[0] == col_table.dtype[0]
assert col_result.size == col_table.size
assert col_result.offset == col_table.offset

desc_cat_table = col_result.describe_categorical
desc_cat_result = col_result.describe_categorical

assert desc_cat_table["is_ordered"] == desc_cat_result["is_ordered"]
assert desc_cat_table["is_dictionary"] == desc_cat_result["is_dictionary"]
assert isinstance(desc_cat_result["categories"]._col, pa.Array)


@pytest.mark.pandas
@pytest.mark.parametrize(
"uint", [pa.uint8(), pa.uint16(), pa.uint32()]
Expand All@@ -170,6 +123,7 @@ def test_pandas_roundtrip(uint, int, float, np_float):
"a": pa.array(arr, type=uint),
"b": pa.array(arr, type=int),
"c": pa.array(np.array(arr, dtype=np_float), type=float),
"d": [True, False, True],
}
)
from pandas.api.interchange import (
Expand All@@ -189,10 +143,10 @@ def test_pandas_roundtrip(uint, int, float, np_float):


@pytest.mark.pandas
def test_roundtrip_pandas_string():
def test_pandas_roundtrip_string():
# See https://github.com/pandas-dev/pandas/issues/50554
if Version(pd.__version__) < Version("1.6"):
pytest.skip("Column.size() called as a method in pandas 2.0.0")
pytest.skip("Column.size() bug in pandas")
Comment thread
AlenkaF marked this conversation as resolved.

arr = ["a", "", "c"]
table = pa.table({"a": pa.array(arr)})
Expand All@@ -218,10 +172,10 @@ def test_roundtrip_pandas_string():


@pytest.mark.pandas
def test_roundtrip_pandas_large_string():
def test_pandas_roundtrip_large_string():
# See https://github.com/pandas-dev/pandas/issues/50554
if Version(pd.__version__) < Version("1.6"):
pytest.skip("Column.size() called as a method in pandas 2.0.0")
pytest.skip("Column.size() bug in pandas")

arr = ["a", "", "c"]
table = pa.table({"a_large": pa.array(arr, type=pa.large_string())})
Expand DownExpand Up@@ -255,10 +209,10 @@ def test_roundtrip_pandas_large_string():


@pytest.mark.pandas
def test_roundtrip_pandas_string_with_missing():
def test_pandas_roundtrip_string_with_missing():
# See https://github.com/pandas-dev/pandas/issues/50554
if Version(pd.__version__) < Version("1.6"):
pytest.skip("Column.size() called as a method in pandas 2.0.0")
pytest.skip("Column.size() bug in pandas")

arr = ["a", "", "c", None]
table = pa.table({"a": pa.array(arr),
Expand DownExpand Up@@ -287,19 +241,28 @@ def test_roundtrip_pandas_string_with_missing():


@pytest.mark.pandas
def test_roundtrip_pandas_boolean():
if Version(pd.__version__) < Version("1.5.0"):
pytest.skip("__dataframe__ added to pandas in 1.5.0")
def test_pandas_roundtrip_categorical():
if Version(pd.__version__) < Version("2.0.2"):

Copy link
Copy Markdown
Member

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

Should the pandas version for skipping tests due to Column.size() bug in pandas match between tests test_pandas_roundtrip_string (v2.0.1) and test_pandas_roundtrip_categorical (v2.0.2)?

Copy link
Copy Markdown
MemberAuthor

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

Good catch! =) There are also bitmasks involved here, so I have to change the skip message 😊

Copy link
Copy Markdown
MemberAuthor

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

I have, unfortunately, added this change in the rebasing process and is therefore not visible in the commit history.
The change can be seen here:

deftest_pandas_roundtrip_categorical():
ifVersion(pd.__version__) <Version("2.0.2"):
pytest.skip("Bitmasks not supported in pandas interchange implementation")

pytest.skip("Bitmasks not supported in pandas interchange implementation")

table = pa.table({"a": [True, False, True]})
arr = ["Mon", "Tue", "Mon", "Wed", "Mon", "Thu", "Fri", "Sat", None]
table = pa.table(
{"weekday": pa.array(arr).dictionary_encode()}
)

from pandas.api.interchange import (
from_dataframe as pandas_from_dataframe
)
pandas_df = pandas_from_dataframe(table)
result = pi.from_dataframe(pandas_df)

assert table.equals(result)
assert result["weekday"].to_pylist() == table["weekday"].to_pylist()
assert pa.types.is_dictionary(table["weekday"].type)
assert pa.types.is_dictionary(result["weekday"].type)
assert pa.types.is_string(table["weekday"].chunk(0).dictionary.type)
assert pa.types.is_large_string(result["weekday"].chunk(0).dictionary.type)
assert pa.types.is_int32(table["weekday"].chunk(0).indices.type)
assert pa.types.is_int8(result["weekday"].chunk(0).indices.type)

table_protocol = table.__dataframe__()
result_protocol = result.__dataframe__()
Expand All@@ -309,10 +272,25 @@ def test_roundtrip_pandas_boolean():
assert table_protocol.num_chunks() == result_protocol.num_chunks()
assert table_protocol.column_names() == result_protocol.column_names()

col_table = table_protocol.get_column(0)
col_result = result_protocol.get_column(0)

assert col_result.dtype[0] == DtypeKind.CATEGORICAL
assert col_result.dtype[0] == col_table.dtype[0]
assert col_result.size() == col_table.size()
assert col_result.offset == col_table.offset

desc_cat_table = col_result.describe_categorical
desc_cat_result = col_result.describe_categorical

assert desc_cat_table["is_ordered"] == desc_cat_result["is_ordered"]
assert desc_cat_table["is_dictionary"] == desc_cat_result["is_dictionary"]
assert isinstance(desc_cat_result["categories"]._col, pa.Array)


@pytest.mark.pandas
@pytest.mark.parametrize("unit", ['s', 'ms', 'us', 'ns'])
def test_roundtrip_pandas_datetime(unit):
def test_pandas_roundtrip_datetime(unit):
if Version(pd.__version__) < Version("1.5.0"):
pytest.skip("__dataframe__ added to pandas in 1.5.0")
from datetime import datetime as dt
Expand DownExpand Up@@ -384,45 +362,6 @@ def test_pandas_to_pyarrow_float16_with_missing():
pi.from_dataframe(df)


@pytest.mark.pandas
def test_pandas_to_pyarrow_string_with_missing():
if Version(pd.__version__) < Version("1.5.0"):
pytest.skip("__dataframe__ added to pandas in 1.5.0")

# pandas is using int64 offsets for string dtype so the constructed
# pyarrow string column will always be a large_string data type
arr = {
"Y": ["a", "b", None], # bool, ColumnNullType.USE_BYTEMASK,
}
df = pd.DataFrame(arr)
expected = pa.table(arr)
result = pi.from_dataframe(df)

assert result[0].to_pylist() == expected[0].to_pylist()
assert pa.types.is_string(expected[0].type)
assert pa.types.is_large_string(result[0].type)


@pytest.mark.pandas
def test_pandas_to_pyarrow_categorical_with_missing():
if Version(pd.__version__) < Version("1.5.0"):
pytest.skip("__dataframe__ added to pandas in 1.5.0")

arr = ["Mon", "Tue", "Mon", "Wed", "Mon", "Thu", "Fri", "Sat", None]
df = pd.DataFrame(
{"weekday": arr}
)
df = df.astype("category")
result = pi.from_dataframe(df)

expected_dictionary = ["Fri", "Mon", "Sat", "Thu", "Tue", "Wed"]
expected_indices = pa.array([1, 4, 1, 5, 1, 3, 0, 2, None], type=pa.int8())

assert result[0].to_pylist() == arr
assert result[0].chunk(0).dictionary.to_pylist() == expected_dictionary
assert result[0].chunk(0).indices.equals(expected_indices)


@pytest.mark.parametrize(
"uint", [pa.uint8(), pa.uint16(), pa.uint32()]
)
Expand Down