Buckets:
| # /// script | |
| # requires-python = ">=3.11" | |
| # dependencies = ["saturate[hf]>=0.1.2"] | |
| # /// | |
| """Cheap laptop check of the source half of the driver: does ids='id' key rows by | |
| the dataset's own id, and does batched() carry them through intact?""" | |
| import importlib.util, sys | |
| spec = importlib.util.spec_from_file_location("ec", "embed_column.py") | |
| ec = importlib.util.module_from_spec(spec); spec.loader.exec_module(ec) | |
| from saturate import dataset_rows | |
| for col in ["caption", "scene", "events_text"]: | |
| rows = dataset_rows(ec.SOURCE, config="default", split="train", | |
| columns=[col], ids="id") | |
| n_items = 0 | |
| for k, (bid, payload) in enumerate(ec.batched(rows, col)): | |
| assert bid.startswith("b-"), bid | |
| assert len(payload["ids"]) == len(payload["texts"]), payload["ids"] | |
| assert all(isinstance(i, str) and i for i in payload["ids"]) | |
| n_items += len(payload["texts"]) | |
| if k == 0: | |
| print(f"{col}: batch0 id={bid!r} ids[:2]={payload['ids'][:2]} " | |
| f"len={len(payload['texts'])} text0={payload['texts'][0][:60]!r}") | |
| if k == 4: | |
| break | |
| print(f"{col}: 5 batches -> {n_items} items\n") | |
Xet Storage Details
- Size:
- 1.19 kB
- Xet hash:
- e6759014b11434f1acf6aa48ab133b57249db5d828cf21540f326fa4709291c5
·
Xet efficiently stores files, intelligently splitting them into unique chunks and accelerating uploads and downloads. More info.