Upload prepare_metadata.py with huggingface_hub
Browse files- prepare_metadata.py +75 -0
prepare_metadata.py
ADDED
|
@@ -0,0 +1,75 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
"""
|
| 2 |
+
Generates metadata CSVs for the DeepFashion retrieval project.
|
| 3 |
+
|
| 4 |
+
Before running:
|
| 5 |
+
1. Download the DeepFashion (In-shop Clothes Retrieval) high-res images
|
| 6 |
+
and place them locally, preserving the original folder structure:
|
| 7 |
+
<DATASET_PATH>/<gender>/<clothing_category>/<item_id>/<image>.jpg
|
| 8 |
+
2. Update DATASET_PATH below to point to your local copy.
|
| 9 |
+
|
| 10 |
+
Outputs (written to the current directory):
|
| 11 |
+
- original_metadata.csv all images found
|
| 12 |
+
- full_metadata.csv with segmentation images removed
|
| 13 |
+
- original_metadata_filtered.csv with known bad/corrupted item_ids removed
|
| 14 |
+
"""
|
| 15 |
+
|
| 16 |
+
import os
|
| 17 |
+
import pandas as pd
|
| 18 |
+
|
| 19 |
+
# ---- CONFIG: update this to your local dataset path ----
|
| 20 |
+
DATASET_PATH = "/mnt/c/Users/User/Downloads/img_highres"
|
| 21 |
+
|
| 22 |
+
# item_ids excluded due to corrupted/mislabeled images found during data cleaning
|
| 23 |
+
EXCLUDED_ITEM_IDS = [
|
| 24 |
+
"id_00003615", "id_00006951", "id_00004776", "id_00004850",
|
| 25 |
+
"id_00007573", "id_00000773", "id_00006128", "id_00006574",
|
| 26 |
+
"id_00001453", "id_00001995", "id_00002205", "id_00003020",
|
| 27 |
+
]
|
| 28 |
+
|
| 29 |
+
|
| 30 |
+
def build_metadata(dataset_path: str) -> pd.DataFrame:
|
| 31 |
+
data = []
|
| 32 |
+
for gender in os.listdir(dataset_path):
|
| 33 |
+
gender_path = os.path.join(dataset_path, gender)
|
| 34 |
+
if not os.path.isdir(gender_path):
|
| 35 |
+
continue
|
| 36 |
+
|
| 37 |
+
for clothing_category in os.listdir(gender_path):
|
| 38 |
+
clothing_cat_path = os.path.join(gender_path, clothing_category)
|
| 39 |
+
if not os.path.isdir(clothing_cat_path):
|
| 40 |
+
continue
|
| 41 |
+
|
| 42 |
+
for item_id in os.listdir(clothing_cat_path):
|
| 43 |
+
item_id_path = os.path.join(clothing_cat_path, item_id)
|
| 44 |
+
if not os.path.isdir(item_id_path):
|
| 45 |
+
continue
|
| 46 |
+
|
| 47 |
+
for image in os.listdir(item_id_path):
|
| 48 |
+
image_path = os.path.join(item_id_path, image)
|
| 49 |
+
data.append({
|
| 50 |
+
"gender": gender,
|
| 51 |
+
"clothing_category": clothing_category,
|
| 52 |
+
"item_id": item_id,
|
| 53 |
+
"image_path": image_path,
|
| 54 |
+
})
|
| 55 |
+
|
| 56 |
+
return pd.DataFrame(data)
|
| 57 |
+
|
| 58 |
+
|
| 59 |
+
def main():
|
| 60 |
+
print(f"Scanning dataset at: {DATASET_PATH}")
|
| 61 |
+
original = build_metadata(DATASET_PATH)
|
| 62 |
+
original.to_csv("original_metadata.csv", index=False)
|
| 63 |
+
print(f"original_metadata.csv written ({len(original)} rows)")
|
| 64 |
+
|
| 65 |
+
full = original[~original.image_path.str.contains("segment")]
|
| 66 |
+
full.to_csv("full_metadata.csv", index=False)
|
| 67 |
+
print(f"full_metadata.csv written ({len(full)} rows)")
|
| 68 |
+
|
| 69 |
+
filtered = full[~full.item_id.isin(EXCLUDED_ITEM_IDS)]
|
| 70 |
+
filtered.to_csv("original_metadata_filtered.csv", index=False)
|
| 71 |
+
print(f"original_metadata_filtered.csv written ({len(filtered)} rows)")
|
| 72 |
+
|
| 73 |
+
|
| 74 |
+
if __name__ == "__main__":
|
| 75 |
+
main()
|