Skip to content

Commit 46d4213

Browse files
committed
feat: read BigLake Lakehouse tables through tables.get instead of pyiceberg
Parse 4-part Lakehouse table IDs (project.catalog.namespace.table) with TableReference.from_string() and load them through tables.get, the same as regular tables, instead of the pyiceberg REST catalog workaround. 4-part support in from_string() requires google-cloud-bigquery>=3.42.0. The BigQuery Storage Read API accepts snapshot_time for Lakehouse tables but returns current data, while SQL FOR SYSTEM_TIME AS OF is honored. Reads pinned to a snapshot time are therefore left to the SQL path so a DataFrame keeps returning the same data after the table changes. Bug filed with the BigQuery team - b/568786565. System tests will be added in a follow-up change. BUG=567983672 Change-Id: I7f1b23d584e0e3e3dc29f1ff63e8654870c00b0e Reviewed-on: https://bigframes-internal-review.git.corp.google.com/c/bigframes/+/4520 Reviewed-by: Tim Swena <swast@google.com> Kokoro-Docs: Kokoro <noreply+kokoro-dedicatedkokoro-dedicated Kokorogoogle.com> Kokoro-E2E: Kokoro <noreply+kokoro-dedicatedkokoro-dedicated Kokorogoogle.com> Kokoro-Lint: Kokoro <noreply+kokoro-dedicatedkokoro-dedicated Kokorogoogle.com> Kokoro-Doctest: Kokoro <noreply+kokoro-dedicatedkokoro-dedicated Kokorogoogle.com> Kokoro-Unit: Kokoro <noreply+kokoro-dedicatedkokoro-dedicated Kokorogoogle.com> Kokoro-Prerelease: Kokoro <noreply+kokoro-dedicatedkokoro-dedicated Kokorogoogle.com> Reviewed-by: Shenyang Cai <sycai@google.com> Kokoro-System: Kokoro <noreply+kokoro-dedicatedkokoro-dedicated Kokorogoogle.com>
1 parent c54ec01 commit 46d4213

16 files changed

Lines changed: 344 additions & 24 deletions

File tree

‎.pre-commit-config.yaml‎

Lines changed: 1 addition & 1 deletion
Original file line numberDiff line numberDiff line change
@@ -61,7 +61,7 @@ repos:
6161
- geopandas>=0.12.2
6262
- google-cloud-bigquery-connection>=1.18.2
6363
- google-cloud-bigquery-storage>=2.30.0,<3.0.0
64-
- google-cloud-bigquery[bqstorage,pandas]>=3.36.0
64+
- google-cloud-bigquery[bqstorage,pandas]>=3.42.0
6565
- google-cloud-functions>=1.20.2
6666
- google-cloud-resource-manager>=1.14.2
6767
- google-cloud-storage>=2.0.0

‎bigframes/core/bq_data.py‎

Lines changed: 28 additions & 0 deletions
Original file line numberDiff line numberDiff line change
@@ -17,6 +17,7 @@
1717
import concurrent.futures
1818
import dataclasses
1919
import datetime
20+
import enum
2021
import functools
2122
import os
2223
import queue
@@ -103,6 +104,22 @@ class TableMetadata:
103104
modified_time: Optional[datetime.datetime] = None
104105

105106

107+
class TableKind(enum.Enum):
108+
"""Where a table comes from.
109+
110+
This is separate from TableMetadata.type, which is the BigQuery table type,
111+
such as TABLE or VIEW.
112+
"""
113+
114+
# A table or view in a BigQuery dataset: project.dataset.table.
115+
NATIVE = enum.auto()
116+
# A BigLake Lakehouse table: project.catalog.namespace.table.
117+
LAKEHOUSE = enum.auto()
118+
# An INFORMATION_SCHEMA view, such as
119+
# project.region-us.INFORMATION_SCHEMA.SCHEMATA.
120+
INFORMATION_SCHEMA = enum.auto()
121+
122+
106123
@dataclasses.dataclass(frozen=True)
107124
class GbqNativeTable:
108125
project_id: str = dataclasses.field()
@@ -170,6 +187,17 @@ def from_ref_and_schema(
170187
def is_physically_stored(self) -> bool:
171188
return self.metadata.type in ["TABLE", "MATERIALIZED_VIEW"]
172189

190+
@property
191+
def kind(self) -> TableKind:
192+
# INFORMATION_SCHEMA views are loaded with INFORMATION_SCHEMA as the
193+
# dataset ID.
194+
if self.dataset_id.casefold() == "INFORMATION_SCHEMA".casefold():
195+
return TableKind.INFORMATION_SCHEMA
196+
# Lakehouse tables have catalog.namespace in place of a dataset ID.
197+
if "." in self.dataset_id:
198+
return TableKind.LAKEHOUSE
199+
return TableKind.NATIVE
200+
173201
def get_table_ref(self) -> bq.TableReference:
174202
return bq.TableReference(
175203
bq.DatasetReference(self.project_id, self.dataset_id), self.table_id

‎bigframes/core/pyformat.py‎

Lines changed: 4 additions & 1 deletion
Original file line numberDiff line numberDiff line change
@@ -115,7 +115,10 @@ def is_biglake(
115115
node: nodes.BigFrameNode, child_results: Tuple[bool, ...]
116116
) -> bool:
117117
if isinstance(node, nodes.ReadTableNode):
118-
return isinstance(node.source.table, bq_data.BiglakeIcebergTable)
118+
return isinstance(node.source.table, bq_data.BiglakeIcebergTable) or (
119+
isinstance(node.source.table, bq_data.GbqNativeTable)
120+
and node.source.table.kind == bq_data.TableKind.LAKEHOUSE
121+
)
119122
return any(child_results)
120123

121124
contains_biglake = value._block.expr.node.reduce_up(is_biglake)

‎bigframes/pandas/io/api.py‎

Lines changed: 0 additions & 9 deletions
Original file line numberDiff line numberDiff line change
@@ -58,9 +58,7 @@
5858
import bigframes.session
5959
import bigframes.session._io.bigquery
6060
import bigframes.session.clients
61-
import bigframes.session.iceberg
6261
import bigframes.session.metrics
63-
from bigframes.core import bq_data
6462
from bigframes.session import dry_runs
6563

6664
# Note: the following methods are duplicated from Session. This duplication
@@ -728,13 +726,6 @@ def _set_default_session_location_if_possible_deferred_query(create_query):
728726
default_project=default_project,
729727
)
730728
config.options.bigquery.location = table.location
731-
elif bq_data.is_irc_table(query):
732-
irc_table = bigframes.session.iceberg.get_table(
733-
default_project, query, bqclient._credentials
734-
)
735-
config.options.bigquery.location = bq_data.get_default_bq_region(
736-
irc_table.metadata.location
737-
)
738729
else:
739730
table = bqclient.get_table(query)
740731
config.options.bigquery.location = table.location

‎bigframes/session/loader.py‎

Lines changed: 0 additions & 5 deletions
Original file line numberDiff line numberDiff line change
@@ -73,7 +73,6 @@
7373
import bigframes.session._io.bigquery as bf_io_bigquery
7474
import bigframes.session._io.bigquery.read_gbq_query as bf_read_gbq_query
7575
import bigframes.session._io.bigquery.read_gbq_table as bf_read_gbq_table
76-
import bigframes.session.iceberg
7776
import bigframes.session.metrics
7877
import bigframes.session.temporary_storage
7978
import bigframes.session.time as session_time
@@ -1084,10 +1083,6 @@ def _get_table_metadata(
10841083
default_project=default_project,
10851084
)
10861085
table = bq_data.GbqNativeTable.from_table(client_table)
1087-
elif bq_data.is_irc_table(table_id):
1088-
table = bigframes.session.iceberg.get_table(
1089-
self._bqclient.project, table_id, self._bqclient._credentials
1090-
)
10911086
else:
10921087
table_ref = google.cloud.bigquery.table.TableReference.from_string(
10931088
table_id, default_project=default_project

‎bigframes/session/read_api_execution.py‎

Lines changed: 9 additions & 0 deletions
Original file line numberDiff line numberDiff line change
@@ -55,6 +55,15 @@ async def execute(
5555
if not node.source.table.is_physically_stored:
5656
return None
5757

58+
# TODO(b/568786565): Remove this branch once the Read API honors
59+
# snapshot_time for Lakehouse tables. Until then, time-travel reads of
60+
# these tables must use SQL, since the Read API returns current data.
61+
if (
62+
node.source.at_time is not None
63+
and node.source.table.kind == bq_data.TableKind.LAKEHOUSE
64+
):
65+
return None
66+
5867
peek = execution_spec.peek
5968
if limit is not None:
6069
if peek is None or limit < peek:

‎setup.py‎

Lines changed: 1 addition & 1 deletion
Original file line numberDiff line numberDiff line change
@@ -39,7 +39,7 @@
3939
"gcsfs >=2023.3.0, !=2025.5.0, !=2026.2.0, !=2026.3.0",
4040
"geopandas >=0.12.2",
4141
"google-auth[pyopenssl] >=2.15.0,<3.0",
42-
"google-cloud-bigquery[bqstorage,pandas] >=3.36.0",
42+
"google-cloud-bigquery[bqstorage,pandas] >=3.42.0",
4343
# 2.30 needed for arrow support.
4444
"google-cloud-bigquery-storage >= 2.30.0, < 3.0.0",
4545
"google-cloud-functions >=1.20.2",

‎testing/constraints-3.10.txt‎

Lines changed: 1 addition & 1 deletion
Original file line numberDiff line numberDiff line change
@@ -6,7 +6,7 @@ geopandas==0.12.2
66
google-auth==2.15.0
77
google-cloud-bigtable==2.30.0
88
google-cloud-pubsub==2.29.0
9-
google-cloud-bigquery==3.36.0
9+
google-cloud-bigquery==3.42.0
1010
google-cloud-functions==1.20.2
1111
google-cloud-bigquery-connection==1.18.2
1212
google-cloud-iam==2.18.2

‎testing/constraints-3.11.txt‎

Lines changed: 1 addition & 1 deletion
Original file line numberDiff line numberDiff line change
@@ -152,7 +152,7 @@ google-auth==2.38.0
152152
google-auth-httplib2==0.2.0
153153
google-auth-oauthlib==1.2.2
154154
google-cloud-aiplatform==1.106.0
155-
google-cloud-bigquery==3.36.0
155+
google-cloud-bigquery==3.42.0
156156
google-cloud-bigquery-connection==1.18.3
157157
google-cloud-bigquery-storage==2.32.0
158158
google-cloud-core==2.4.3
Original file line numberDiff line numberDiff line change
@@ -0,0 +1,3 @@
1+
SELECT
2+
*
3+
FROM `bigframes-dev`.`my_catalog.my_namespace`.`scalar_types` AS `bft_0` FOR SYSTEM_TIME AS OF '2025-11-09T03:04:05.678901+00:00'

0 commit comments

Comments
 (0)