-
Notifications
You must be signed in to change notification settings - Fork 0
Expand file tree
/
Copy pathimage_dataset.py
More file actions
96 lines (77 loc) · 3.38 KB
/
Copy pathimage_dataset.py
File metadata and controls
96 lines (77 loc) · 3.38 KB
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
"""
Build an image dataset with source attribution.
export CHOCODATA_API_KEY="your_key"
python bing_image_scraper_api_codes/image_dataset.py "red panda" "bonsai tree"
For each query it pulls the image grid, keeps one row per distinct full-size image
URL, and writes a CSV manifest carrying the source page and host next to every
image so the dataset stays attributable after the fact. Add --download to also
fetch the image bytes.
The manifest columns are:
query, position, title, image, thumbnail, source_page, source, file
`source_page` and `source` are the reason to keep a manifest at all: an image URL
on its own tells you nothing about where it came from or who to credit.
"""
import argparse
import csv
import os
import pathlib
import sys
import time
import requests
from images import images
OUT = pathlib.Path("bing_image_dataset")
def safe_name(url: str, index: int) -> str:
tail = url.split("?")[0].split("/")[-1][:60]
if "." not in tail:
tail = f"{tail or 'image'}.jpg"
return f"{index:04d}_{tail}"
def build(queries: list[str], per_query: int, download: bool) -> None:
OUT.mkdir(exist_ok=True)
manifest = OUT / "manifest.csv"
seen: set[str] = set()
written = 0
with manifest.open("w", newline="", encoding="utf-8") as fh:
w = csv.writer(fh)
w.writerow(["query", "position", "title", "image", "thumbnail",
"source_page", "source", "file"])
for q in queries:
data = images(q, count=per_query)
rows = data.get("results", [])
print(f"{q!r}: {data['results_count']} images "
f"({len({r['source'] for r in rows})} source domains)")
for row in rows:
url = row.get("image")
if not url or url in seen:
continue
seen.add(url)
fname = ""
if download:
fname = safe_name(url, written)
try:
# Publisher hosts are third-party; time out and move on.
img = requests.get(url, timeout=20)
if img.ok and img.content:
(OUT / fname).write_bytes(img.content)
else:
fname = ""
except requests.RequestException:
fname = ""
time.sleep(0.2)
w.writerow([q, row["position"], row["title"], url,
row.get("thumbnail"), row.get("source_page"),
row.get("source"), fname])
written += 1
print()
print(f"{written} unique images across {len(queries)} quer"
f"{'y' if len(queries) == 1 else 'ies'} -> {manifest}")
if not download:
print("Re-run with --download to also fetch the image bytes.")
if __name__ == "__main__":
ap = argparse.ArgumentParser(description="Build an attributed Bing image dataset")
ap.add_argument("queries", nargs="*", default=["red panda"])
ap.add_argument("--per-query", type=int, default=20,
help="upper bound per query; Bing decides how many it renders")
ap.add_argument("--download", action="store_true",
help="also download the full-size image bytes")
args = ap.parse_args()
build(args.queries or ["red panda"], args.per_query, args.download)