Buckets:
| """Build sorted parquet index copies so DuckDB zone-map pruning makes lookups fast. | |
| Unsorted 104GB parquet forces a full scan (~90s) per query. A copy sorted by the | |
| search key lets DuckDB skip all but the matching row groups (ms-level lookups). | |
| Creates (skips files that already exist -> resumable): | |
| idx_phone.parquet - sorted by phoneNumber | |
| idx_aadhar.parquet - sorted by aadharNumber | |
| idx_name.parquet - sorted by name | |
| """ | |
| import duckdb | |
| import os | |
| import time | |
| FILES = ["part1.parquet", "part2a.parquet", "part2b_new.parquet"] | |
| FILES_SQL = ", ".join(f"'{f}'" for f in FILES) | |
| def build(con, field: str, out: str): | |
| if os.path.exists(out) and os.path.getsize(out) > 0: | |
| print(f"{out} already exists, skipping", flush=True) | |
| return | |
| print(f"START {out} (sort by {field})", flush=True) | |
| t = time.time() | |
| con.execute(f""" | |
| COPY (SELECT * FROM people ORDER BY {field} NULLS LAST) | |
| TO '{out}' (FORMAT PARQUET, COMPRESSION ZSTD) | |
| """) | |
| print(f"DONE {out} in {time.time()-t:.1f}s", flush=True) | |
| con = duckdb.connect() | |
| con.execute("INSTALL parquet; LOAD parquet;") | |
| con.execute("SET threads = 12") | |
| con.execute("SET memory_limit = '27GB'") | |
| con.execute(f"CREATE VIEW people AS SELECT * FROM read_parquet([{FILES_SQL}])") | |
| build(con, "phoneNumber", "idx_phone.parquet") | |
| build(con, "aadharNumber", "idx_aadhar.parquet") | |
| build(con, "name", "idx_name.parquet") | |
| print("ALL DONE", flush=True) | |
Xet Storage Details
- Size:
- 1.43 kB
- Xet hash:
- c9b6227b85bfb8dc71f02b3f554a4180de3fae2dfb5ae21824ef1e74a194ed69
·
Xet efficiently stores files, intelligently splitting them into unique chunks and accelerating uploads and downloads. More info.