Repository navigation
Expand file tree
/
Copy pathdataset_probe.py
More file actions
87 lines (76 loc) · 3.15 KB
/
Copy pathdataset_probe.py
File metadata and controls
87 lines (76 loc) · 3.15 KB
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
import json
import os
from pathlib import Path
import torch
from PIL import Image
from torch.utils.data import Dataset
from torchvision import transforms
DATA_ROOT = Path(os.environ.get("INPUT_DATA", "./data"))
SAT_ROOT = DATA_ROOT / "satellite_image"
VALIDATED_DATASET = DATA_ROOT / "validated" / "Dataset"
VALIDATED_BENCHMARK = DATA_ROOT / "validated" / "Benchmark"
IMG_MEAN = (0.485, 0.456, 0.406)
IMG_STD = (0.229, 0.224, 0.225)
def eval_transform(size: int = 224):
return transforms.Compose([
transforms.Resize(size),
transforms.CenterCrop(size),
transforms.ToTensor(),
transforms.Normalize(IMG_MEAN, IMG_STD),
])
def _sat_path_from_validated(p: str) -> Path:
parts = p.split("/satellite_image/")
if len(parts) != 2:
return Path("")
return SAT_ROOT / parts[1]
class CityLensDownstream(Dataset):
BENCHMARK_FILES = {
"pop": "all_global_pop_task.json",
"gdp": "all_global_gdp_task.json",
"build_height": "all_global_build_height_task.json",
"acc2health": "all_global_acc2health_task.json",
"house_price": "all_house_price_task.json",
"life_exp": "UK_life_expectancy_task.json",
"bachelor": "US_bachelor_ratio_task.json",
"mental": "US_health_mental_task.json",
"drive": "US_transport_drive_task.json",
"public": "US_transport_public_task.json",
}
ALL_FILES = {
"pop": "all_global_pop_task_all.json",
"gdp": "all_global_gdp_task_all.json",
"build_height": "all_global_build_height_task-all.json",
"acc2health": "all_global_acc2health_task-all.json",
"house_price": "all_house_price_task-all.json",
"life_exp": "UK_life_expectancy_task-all.json",
"bachelor": "US_bachelor_ratio_task-all.json",
"mental": "US_health_mental_task-all.json",
"drive": "US_transport_drive_task-all.json",
"public": "US_transport_public_task-all.json",
}
TASKS = list(BENCHMARK_FILES.keys())
def __init__(self, task: str, transform=None, target: str = "reference", split: str = "all"):
assert task in self.TASKS, f"unknown task {task}; choose {self.TASKS}"
assert split in ("benchmark", "all")
root = VALIDATED_BENCHMARK if split == "benchmark" else VALIDATED_DATASET
fname = (self.BENCHMARK_FILES if split == "benchmark" else self.ALL_FILES)[task]
with open(root / fname) as f:
records = json.load(f)
self.samples = []
for r in records:
if not r.get("images"):
continue
sat = _sat_path_from_validated(r["images"][0])
if not sat.exists():
continue
y = r.get(target, r.get("reference"))
if y is not None:
self.samples.append((sat, float(y)))
self.transform = transform or eval_transform()
print(f"==== Num of probing samples: {len(self.samples)} ====")
def __len__(self):
return len(self.samples)
def __getitem__(self, i):
img_path, y = self.samples[i]
img = Image.open(img_path).convert("RGB")
return self.transform(img), torch.tensor(y, dtype=torch.float32)