Download src/utils/extract_subset.py from thanhhuyvan/Gaze-LIPE: direct link, hf CLI and curl.
- Browser
- Download file 2.81 kB
-
https://huggingface.co/thanhhuyvan/Gaze-LIPE/resolve/main/src/utils/extract_subset.py
- Command line
-
hf download hf://thanhhuyvan/Gaze-LIPE/src/utils/extract_subset.py
-
curl -L -o extract_subset.py https://huggingface.co/thanhhuyvan/Gaze-LIPE/resolve/main/src/utils/extract_subset.py
2.81 kB
| import os | |
| import tarfile | |
| import io | |
| def join_and_extract(parts_dir, output_dir, target_rec='rec_000'): | |
| """ | |
| Joins tar parts and extracts only the target recording to save space. | |
| """ | |
| parts = sorted([f for f in os.listdir(parts_dir) if 'Gaze360.tar.part' in f]) | |
| if not parts: | |
| print("No tar parts found!") | |
| return | |
| print(f"Found parts: {parts}") | |
| os.makedirs(output_dir, exist_ok=True) | |
| # We use a custom stream class to join parts without writing a huge file to disk | |
| class JoinedStream(io.RawIOBase): | |
| def __init__(self, parts_dir, parts): | |
| self.parts_dir = parts_dir | |
| self.parts = parts | |
| self.current_part_idx = 0 | |
| self.current_file = open(os.path.join(parts_dir, parts[0]), 'rb') | |
| def readinto(self, b): | |
| n = self.current_file.readinto(b) | |
| if n == 0 and self.current_part_idx < len(self.parts) - 1: | |
| self.current_file.close() | |
| self.current_part_idx += 1 | |
| print(f"Switching to {self.parts[self.current_part_idx]}...") | |
| self.current_file = open(os.path.join(self.parts_dir, self.parts[self.current_part_idx]), 'rb') | |
| n = self.current_file.readinto(b) | |
| return n | |
| def readable(self): | |
| return True | |
| print(f"Opening joined stream and searching for {target_rec}...") | |
| stream = JoinedStream(parts_dir, parts) | |
| try: | |
| with tarfile.open(fileobj=stream, mode='r|') as tar: | |
| for member in tar: | |
| # Gaze360 tar structure usually has imgs/rec_xxx/head/... | |
| if target_rec in member.name and member.isfile(): | |
| # Flatten the path a bit for our local structure: data/raw/imgs/rec_xxx/... | |
| # Original might be Gaze360/imgs/rec_000/... | |
| parts = member.name.split('/') | |
| if 'imgs' in parts: | |
| idx = parts.index('imgs') | |
| rel_path = os.path.join(*parts[idx+1:]) | |
| target_path = os.path.join(output_dir, rel_path) | |
| os.makedirs(os.path.dirname(target_path), exist_ok=True) | |
| with open(target_path, 'wb') as f: | |
| f.write(tar.extractfile(member).read()) | |
| # Optional: limit extraction to first 500 images of the recording for speed/space | |
| # if "head" in member.name and "000500.jpg" in member.name: break | |
| except Exception as e: | |
| print(f"Note: Stream ended or error occurred: {e}") | |
| finally: | |
| stream.current_file.close() | |
| if __name__ == "__main__": | |
| join_and_extract('data/raw', 'data/raw/imgs', target_rec='rec_000') | |