Skip to content

Commit

Permalink
feat: add example of reading parquet from s3
Browse files Browse the repository at this point in the history
  • Loading branch information
mesejo committed Aug 21, 2023
1 parent 217ede8 commit f84c829
Showing 1 changed file with 22 additions and 0 deletions.
22 changes: 22 additions & 0 deletions examples/sql-parquet-s3.py
Original file line number Diff line number Diff line change
@@ -0,0 +1,22 @@
import os
import datafusion
from datafusion.object_store import AmazonS3

region = "us-east-1"
bucket_name = "yellow-trips"

s3 = AmazonS3(
bucket_name=bucket_name,
region=region,
access_key_id=os.getenv("AWS_ACCESS_KEY_ID"),
secret_access_key=os.getenv("AWS_SECRET_ACCESS_KEY"),
)

ctx = datafusion.SessionContext()
path = f"s3://{bucket_name}/"
ctx.register_object_store(path, s3)

ctx.register_parquet("trips", path)

df = ctx.sql("select count(passenger_count) from trips")
df.show()

0 comments on commit f84c829

Please sign in to comment.