# Copyright 2026 Google LLC
#
# Licensed under the Apache License, Version 2.0 (the "License");
# you may not use this file except in compliance with the License.
# You may obtain a copy of the License at
#
# https://www.apache.org/licenses/LICENSE-2.0
#
# Unless required by applicable law or agreed to in writing, software
# distributed under the License is distributed on an "AS IS" BASIS,
# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
# See the License for the specific language governing permissions and
# limitations under the License.
Use AI Functions with the BigQuery Accessor#
Run in Colab
|
|
|
|
Environment Setup#
Make sure your GCP project have the follwing roles:
Then import bigframes to enable the bigquery accessor:
import bigframes.pandas as bpd
PROJECT_ID = "" # @param {type:"string"}
LOCATION = "US" # @param {type:"string"}
bpd.options.bigquery.project = PROJECT_ID
bpd.options.bigquery.location = LOCATION
bpd.options.display.progress_bar = None
Functions on pandas Series#
Example: AI.EMBED#
import pandas as pd
animals = pd.Series(['dog', 'fish'])
result = animals.bigquery.ai.embed(endpoint='text-embedding-005')
display(result)
print(f"The type of the result is: {type(result)}")
0 {'result': array([ 1.78243860e-03, -1.10658340...
1 {'result': array([-7.29714287e-03, 1.04725976...
dtype: struct<result: list<item: double>, status: string>[pyarrow]
The type of the result is: <class 'pandas.Series'>
Example: AI.SIMILARITY#
Computes similarities between a series and a constant:
import pandas as pd
animals = pd.Series(['dog', 'fish'])
animals.bigquery.ai.similarity('shrimp', endpoint='text-embedding-005')
0 0.629751
1 0.768028
dtype: Float64
Computes similiarities between two series:
import pandas as pd
animals = pd.Series(['dog', 'fish'])
random_stuff = pd.Series(['smart phone', 'salmon'])
animals.bigquery.ai.similarity(random_stuff, endpoint='text-embedding-005')
0 0.651206
1 0.835251
dtype: Float64
Functions on pandas DataFrames#
Example: AI.PREDICT#
First, prepare a dataset suitable for regression:
import pandas as pd
train_df = pd.DataFrame({
"neighborhood": [
"Downtown", "Downtown", "Downtown", "Downtown",
"Suburbs", "Suburbs", "Suburbs", "Suburbs",
"Midtown", "Midtown", "Midtown", "Midtown",
],
"bedrooms": [1, 2, 2, 3, 2, 3, 3, 4, 1, 2, 2, 3],
"bathrooms": [1.0, 1.5, 2.0, 2.5, 1.5, 2.0, 2.5, 3.0, 1.0, 1.5, 2.0, 2.5],
"sqft": [650, 900, 1100, 1500, 950, 1300, 1600, 2100, 700, 950, 1200, 1650],
"monthly_rent": [2200, 2800, 3300, 4200, 1600, 2100, 2500, 3100, 1950, 2500, 3000, 3800],
})
predict_df = pd.DataFrame({
"neighborhood": ["Downtown", "Suburbs", "Midtown"],
"bedrooms": [2, 3, 1],
"bathrooms": [2.0, 2.0, 1.5],
"sqft": [1050, 1450, 850],
})
print("=====Training data:")
display(train_df)
print("=====Prediction data:")
display(predict_df)
=====Training data:
| neighborhood | bedrooms | bathrooms | sqft | monthly_rent | |
|---|---|---|---|---|---|
| 0 | Downtown | 1 | 1.0 | 650 | 2200 |
| 1 | Downtown | 2 | 1.5 | 900 | 2800 |
| 2 | Downtown | 2 | 2.0 | 1100 | 3300 |
| 3 | Downtown | 3 | 2.5 | 1500 | 4200 |
| 4 | Suburbs | 2 | 1.5 | 950 | 1600 |
| 5 | Suburbs | 3 | 2.0 | 1300 | 2100 |
| 6 | Suburbs | 3 | 2.5 | 1600 | 2500 |
| 7 | Suburbs | 4 | 3.0 | 2100 | 3100 |
| 8 | Midtown | 1 | 1.0 | 700 | 1950 |
| 9 | Midtown | 2 | 1.5 | 950 | 2500 |
| 10 | Midtown | 2 | 2.0 | 1200 | 3000 |
| 11 | Midtown | 3 | 2.5 | 1650 | 3800 |
=====Prediction data:
| neighborhood | bedrooms | bathrooms | sqft | |
|---|---|---|---|---|
| 0 | Downtown | 2 | 2.0 | 1050 |
| 1 | Suburbs | 3 | 2.0 | 1450 |
| 2 | Midtown | 1 | 1.5 | 850 |
Then, perform a regression with TabFM without prior model training:
predictions = train_df.bigquery.ai.predict(
predict_df,
label_col="monthly_rent",
)
predictions
| neighborhood | bedrooms | bathrooms | sqft | predicted_monthly_rent | |
|---|---|---|---|---|---|
| 0 | Suburbs | 3 | 2.0 | 1450 | 2008.0 |
| 1 | Midtown | 1 | 1.5 | 850 | 2352.0 |
| 2 | Downtown | 2 | 2.0 | 1050 | 3392.0 |
Example: AI.FORECAST#
First, prepare a timeseries dataset:
import pandas as pd
dates = pd.date_range("2025-01-01", periods=14, freq="D")
df = pd.DataFrame({
"date": dates.tolist() * 2,
"item_id": ["item_1"] * 14 + ["item_2"] * 14,
"sales": [
100, 105, 110, 115, 120, 125, 130, 135, 140, 145, 150, 155, 160, 165,
50, 52, 55, 58, 60, 62, 65, 68, 70, 72, 75, 78, 80, 85,
],
})
df
| date | item_id | sales | |
|---|---|---|---|
| 0 | 2025-01-01 | item_1 | 100 |
| 1 | 2025-01-02 | item_1 | 105 |
| 2 | 2025-01-03 | item_1 | 110 |
| 3 | 2025-01-04 | item_1 | 115 |
| 4 | 2025-01-05 | item_1 | 120 |
| 5 | 2025-01-06 | item_1 | 125 |
| 6 | 2025-01-07 | item_1 | 130 |
| 7 | 2025-01-08 | item_1 | 135 |
| 8 | 2025-01-09 | item_1 | 140 |
| 9 | 2025-01-10 | item_1 | 145 |
| 10 | 2025-01-11 | item_1 | 150 |
| 11 | 2025-01-12 | item_1 | 155 |
| 12 | 2025-01-13 | item_1 | 160 |
| 13 | 2025-01-14 | item_1 | 165 |
| 14 | 2025-01-01 | item_2 | 50 |
| 15 | 2025-01-02 | item_2 | 52 |
| 16 | 2025-01-03 | item_2 | 55 |
| 17 | 2025-01-04 | item_2 | 58 |
| 18 | 2025-01-05 | item_2 | 60 |
| 19 | 2025-01-06 | item_2 | 62 |
| 20 | 2025-01-07 | item_2 | 65 |
| 21 | 2025-01-08 | item_2 | 68 |
| 22 | 2025-01-09 | item_2 | 70 |
| 23 | 2025-01-10 | item_2 | 72 |
| 24 | 2025-01-11 | item_2 | 75 |
| 25 | 2025-01-12 | item_2 | 78 |
| 26 | 2025-01-13 | item_2 | 80 |
| 27 | 2025-01-14 | item_2 | 85 |
Then, use TimesFM to forecase the time series
df.bigquery.ai.forecast(
data_col="sales",
timestamp_col="date",
id_cols=["item_id"],
horizon=3
)
| item_id | forecast_timestamp | forecast_value | confidence_level | prediction_interval_lower_bound | prediction_interval_upper_bound | ai_forecast_status | |
|---|---|---|---|---|---|---|---|
| 0 | item_1 | 2025-01-16 00:00:00+00:00 | 174.257492 | 0.95 | 175.728216 | 177.85742 | |
| 1 | item_1 | 2025-01-15 00:00:00+00:00 | 167.466003 | 0.95 | 169.34789 | 169.553343 | |
| 2 | item_1 | 2025-01-17 00:00:00+00:00 | 180.491333 | 0.95 | 178.587954 | 185.040092 | |
| 3 | item_2 | 2025-01-17 00:00:00+00:00 | 91.99752 | 0.95 | 79.145413 | 100.786089 | |
| 4 | item_2 | 2025-01-15 00:00:00+00:00 | 87.950104 | 0.95 | 82.876698 | 92.470649 | |
| 5 | item_2 | 2025-01-16 00:00:00+00:00 | 90.050346 | 0.95 | 81.460533 | 96.463373 |
Example: AI.GENERATE_BOOL#
Evaluates a structured prompt condition using Gemini and returns a boolean series:
import pandas as pd
df = pd.DataFrame({
"animal": ["cougar", "fish"],
"habitat": ["mountain", "grassland"]
})
prompt = (df["animal"], "lives in the ", df["habitat"])
df.bigquery.ai.generate_bool(prompt, endpoint="gemini-2.5-flash").struct.field("result")
0 True
1 False
Name: result, dtype: bool[pyarrow]
Run in Colab