Uh oh!
There was an error while loading. Please reload this page.
- Notifications
You must be signed in to change notification settings - Fork 68
feat: add DataFrame.struct.explode to add struct subfields to a DataFrame#916
New issue
Have a question about this project? Sign up for a free GitHub account to open an issue and contact its maintainers and the community.
By clicking “Sign up for GitHub”, you agree to our terms of service and privacy statement. We’ll occasionally send you account related emails.
Already on GitHub? Sign in to your account
Uh oh!
There was an error while loading. Please reload this page.
Changes from all commits
c4b5b7f6d2a84e19a397522a47b27db7042File filter
Filter by extension
Conversations
Uh oh!
There was an error while loading. Please reload this page.
Jump to
Uh oh!
There was an error while loading. Please reload this page.
Diff view
Diff view
There are no files selected for viewing
| Original file line number | Diff line number | Diff line change |
|---|---|---|
| @@ -0,0 +1,36 @@ | ||
| # Copyright 2023 Google LLC | ||
| # | ||
| # Licensed under the Apache License, Version 2.0 (the "License"); | ||
| # you may not use this file except in compliance with the License. | ||
| # You may obtain a copy of the License at | ||
| # | ||
| # http://www.apache.org/licenses/LICENSE-2.0 | ||
| # | ||
| # Unless required by applicable law or agreed to in writing, software | ||
| # distributed under the License is distributed on an "AS IS" BASIS, | ||
| # WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. | ||
| # See the License for the specific language governing permissions and | ||
| # limitations under the License. | ||
| """Utility functions for implementing 'explode' functions.""" | ||
| from typing import cast, Sequence, Union | ||
| import bigframes.core.blocks as blocks | ||
| import bigframes.core.utils as utils | ||
| def check_column( | ||
| column: Union[blocks.Label, Sequence[blocks.Label]], | ||
| ) -> Sequence[blocks.Label]: | ||
| if not utils.is_list_like(column): | ||
| column_labels = cast(Sequence[blocks.Label], (column,)) | ||
| else: | ||
| column_labels = cast(Sequence[blocks.Label], tuple(column)) | ||
| if not column_labels: | ||
| raise ValueError("column must be nonempty") | ||
| if len(column_labels) > len(set(column_labels)): | ||
| raise ValueError("column must be unique") | ||
| return column_labels | ||
| Original file line number | Diff line number | Diff line change |
|---|---|---|
| @@ -57,3 +57,26 @@ def dtypes(self) -> pd.Series: | ||
| ], | ||
| index=[pa_type.field(i).name for i in range(pa_type.num_fields)], | ||
| ) | ||
| @log_adapter.class_logger | ||
| class StructFrameAccessor(vendoracessors.StructFrameAccessor): | ||
| __doc__ = vendoracessors.StructAccessor.__doc__ | ||
| def __init__(self, data: bigframes.dataframe.DataFrame) -> None: | ||
| self._parent = data | ||
| def explode(self, column, *, separator: str = ".") -> bigframes.dataframe.DataFrame: | ||
| df = self._parent | ||
| column_labels = bigframes.core.explode.check_column(column) | ||
| for label in column_labels: | ||
CollaboratorAuthor There was a problem hiding this comment. Choose a reason for hiding this commentThe reason will be displayed to describe this comment to others. Learn more. TODO: Add test with multiple columns to explode. CollaboratorAuthor There was a problem hiding this comment. Choose a reason for hiding this commentThe reason will be displayed to describe this comment to others. Learn more. Done in 6d2a84e | ||
| position = df.columns.to_list().index(label) | ||
| df = df.drop(columns=label) | ||
| subfields = self._parent[label].struct.explode() | ||
| for subfield in reversed(subfields.columns): | ||
| df.insert( | ||
| position, f"{label}{separator}{subfield}", subfields[subfield] | ||
| ) | ||
| return df | ||
| Original file line number | Diff line number | Diff line change |
|---|---|---|
| @@ -0,0 +1,40 @@ | ||
| # Copyright 2024 Google LLC | ||
| # | ||
| # Licensed under the Apache License, Version 2.0 (the "License"); | ||
| # you may not use this file except in compliance with the License. | ||
| # You may obtain a copy of the License at | ||
| # | ||
| # http://www.apache.org/licenses/LICENSE-2.0 | ||
| # | ||
| # Unless required by applicable law or agreed to in writing, software | ||
| # distributed under the License is distributed on an "AS IS" BASIS, | ||
| # WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. | ||
| # See the License for the specific language governing permissions and | ||
| # limitations under the License. | ||
| def test_dataframe_struct_explode_multiple_columns(nested_df): | ||
| got = nested_df.struct.explode(["label", "address"]) | ||
| assert got.columns.to_list() == [ | ||
| "customer_id", | ||
| "day", | ||
| "flag", | ||
| "label.key", | ||
| "label.value", | ||
| "event_sequence", | ||
| "address.street", | ||
| "address.city", | ||
| ] | ||
| def test_dataframe_struct_explode_separator(nested_df): | ||
| got = nested_df.struct.explode("label", separator="__sep__") | ||
| assert got.columns.to_list() == [ | ||
| "customer_id", | ||
| "day", | ||
| "flag", | ||
| "label__sep__key", | ||
| "label__sep__value", | ||
| "event_sequence", | ||
| "address", | ||
| ] |
There was a problem hiding this comment.
Choose a reason for hiding this comment
The reason will be displayed to describe this comment to others. Learn more.
Note: Intentionally not 2024 because this code was pulled from a file with date marked as 2023.