ShushanSS commited on
Commit
fe66135
·
verified ·
1 Parent(s): d13852c

Upload prepare_metadata.py with huggingface_hub

Browse files
Files changed (1) hide show
  1. prepare_metadata.py +75 -0
prepare_metadata.py ADDED
@@ -0,0 +1,75 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ """
2
+ Generates metadata CSVs for the DeepFashion retrieval project.
3
+
4
+ Before running:
5
+ 1. Download the DeepFashion (In-shop Clothes Retrieval) high-res images
6
+ and place them locally, preserving the original folder structure:
7
+ <DATASET_PATH>/<gender>/<clothing_category>/<item_id>/<image>.jpg
8
+ 2. Update DATASET_PATH below to point to your local copy.
9
+
10
+ Outputs (written to the current directory):
11
+ - original_metadata.csv all images found
12
+ - full_metadata.csv with segmentation images removed
13
+ - original_metadata_filtered.csv with known bad/corrupted item_ids removed
14
+ """
15
+
16
+ import os
17
+ import pandas as pd
18
+
19
+ # ---- CONFIG: update this to your local dataset path ----
20
+ DATASET_PATH = "/mnt/c/Users/User/Downloads/img_highres"
21
+
22
+ # item_ids excluded due to corrupted/mislabeled images found during data cleaning
23
+ EXCLUDED_ITEM_IDS = [
24
+ "id_00003615", "id_00006951", "id_00004776", "id_00004850",
25
+ "id_00007573", "id_00000773", "id_00006128", "id_00006574",
26
+ "id_00001453", "id_00001995", "id_00002205", "id_00003020",
27
+ ]
28
+
29
+
30
+ def build_metadata(dataset_path: str) -> pd.DataFrame:
31
+ data = []
32
+ for gender in os.listdir(dataset_path):
33
+ gender_path = os.path.join(dataset_path, gender)
34
+ if not os.path.isdir(gender_path):
35
+ continue
36
+
37
+ for clothing_category in os.listdir(gender_path):
38
+ clothing_cat_path = os.path.join(gender_path, clothing_category)
39
+ if not os.path.isdir(clothing_cat_path):
40
+ continue
41
+
42
+ for item_id in os.listdir(clothing_cat_path):
43
+ item_id_path = os.path.join(clothing_cat_path, item_id)
44
+ if not os.path.isdir(item_id_path):
45
+ continue
46
+
47
+ for image in os.listdir(item_id_path):
48
+ image_path = os.path.join(item_id_path, image)
49
+ data.append({
50
+ "gender": gender,
51
+ "clothing_category": clothing_category,
52
+ "item_id": item_id,
53
+ "image_path": image_path,
54
+ })
55
+
56
+ return pd.DataFrame(data)
57
+
58
+
59
+ def main():
60
+ print(f"Scanning dataset at: {DATASET_PATH}")
61
+ original = build_metadata(DATASET_PATH)
62
+ original.to_csv("original_metadata.csv", index=False)
63
+ print(f"original_metadata.csv written ({len(original)} rows)")
64
+
65
+ full = original[~original.image_path.str.contains("segment")]
66
+ full.to_csv("full_metadata.csv", index=False)
67
+ print(f"full_metadata.csv written ({len(full)} rows)")
68
+
69
+ filtered = full[~full.item_id.isin(EXCLUDED_ITEM_IDS)]
70
+ filtered.to_csv("original_metadata_filtered.csv", index=False)
71
+ print(f"original_metadata_filtered.csv written ({len(filtered)} rows)")
72
+
73
+
74
+ if __name__ == "__main__":
75
+ main()