Skip to content

Commit

Permalink
feat: add example of reading parquet from s3
Browse files Browse the repository at this point in the history
  • Loading branch information
mesejo committed Aug 21, 2023
1 parent 217ede8 commit b923b61
Showing 1 changed file with 39 additions and 0 deletions.
39 changes: 39 additions & 0 deletions examples/sql-parquet-s3.py
Original file line number Diff line number Diff line change
@@ -0,0 +1,39 @@
# Licensed to the Apache Software Foundation (ASF) under one
# or more contributor license agreements. See the NOTICE file
# distributed with this work for additional information
# regarding copyright ownership. The ASF licenses this file
# to you under the Apache License, Version 2.0 (the
# "License"); you may not use this file except in compliance
# with the License. You may obtain a copy of the License at
#
# http://www.apache.org/licenses/LICENSE-2.0
#
# Unless required by applicable law or agreed to in writing,
# software distributed under the License is distributed on an
# "AS IS" BASIS, WITHOUT WARRANTIES OR CONDITIONS OF ANY
# KIND, either express or implied. See the License for the
# specific language governing permissions and limitations
# under the License.

import os
import datafusion
from datafusion.object_store import AmazonS3

region = "us-east-1"
bucket_name = "yellow-trips"

s3 = AmazonS3(
bucket_name=bucket_name,
region=region,
access_key_id=os.getenv("AWS_ACCESS_KEY_ID"),
secret_access_key=os.getenv("AWS_SECRET_ACCESS_KEY"),
)

ctx = datafusion.SessionContext()
path = f"s3://{bucket_name}/"
ctx.register_object_store(path, s3)

ctx.register_parquet("trips", path)

df = ctx.sql("select count(passenger_count) from trips")
df.show()

0 comments on commit b923b61

Please sign in to comment.