hashundefined/full / build_index.py
hashundefined's picture
download
raw
1.43 kB
"""Build sorted parquet index copies so DuckDB zone-map pruning makes lookups fast.
Unsorted 104GB parquet forces a full scan (~90s) per query. A copy sorted by the
search key lets DuckDB skip all but the matching row groups (ms-level lookups).
Creates (skips files that already exist -> resumable):
idx_phone.parquet - sorted by phoneNumber
idx_aadhar.parquet - sorted by aadharNumber
idx_name.parquet - sorted by name
"""
import duckdb
import os
import time
FILES = ["part1.parquet", "part2a.parquet", "part2b_new.parquet"]
FILES_SQL = ", ".join(f"'{f}'" for f in FILES)
def build(con, field: str, out: str):
if os.path.exists(out) and os.path.getsize(out) > 0:
print(f"{out} already exists, skipping", flush=True)
return
print(f"START {out} (sort by {field})", flush=True)
t = time.time()
con.execute(f"""
COPY (SELECT * FROM people ORDER BY {field} NULLS LAST)
TO '{out}' (FORMAT PARQUET, COMPRESSION ZSTD)
""")
print(f"DONE {out} in {time.time()-t:.1f}s", flush=True)
con = duckdb.connect()
con.execute("INSTALL parquet; LOAD parquet;")
con.execute("SET threads = 12")
con.execute("SET memory_limit = '27GB'")
con.execute(f"CREATE VIEW people AS SELECT * FROM read_parquet([{FILES_SQL}])")
build(con, "phoneNumber", "idx_phone.parquet")
build(con, "aadharNumber", "idx_aadhar.parquet")
build(con, "name", "idx_name.parquet")
print("ALL DONE", flush=True)

Xet Storage Details

Size:
1.43 kB
·
Xet hash:
c9b6227b85bfb8dc71f02b3f554a4180de3fae2dfb5ae21824ef1e74a194ed69

Xet efficiently stores files, intelligently splitting them into unique chunks and accelerating uploads and downloads. More info.