-
Notifications
You must be signed in to change notification settings - Fork 0
Expand file tree
/
Copy pathcsv_to_json.py
More file actions
212 lines (176 loc) · 6.96 KB
/
Copy pathcsv_to_json.py
File metadata and controls
212 lines (176 loc) · 6.96 KB
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
129
130
131
132
133
134
135
136
137
138
139
140
141
142
143
144
145
146
147
148
149
150
151
152
153
154
155
156
157
158
159
160
161
162
163
164
165
166
167
168
169
170
171
172
173
174
175
176
177
178
179
180
181
182
183
184
185
186
187
188
189
190
191
192
193
194
195
196
197
198
199
200
201
202
203
204
205
206
207
208
209
210
211
212
#!/usr/bin/env python3
import pandas as pd
import datetime
import argparse
import requests
import dotenv
import json
import os
dotenv.load_dotenv()
# Note: Filtering for different columns is not required. Columns matching the names in the schema will be the only ones updated to the database.
def arrayify(
record,
space_sep_fields,
):
"""Converts all space separated values into arrays.
Note that the `primary_decade` field is converted to an array of integers,
not strings.
Args:
record (`pd.Series`): A dataframe record
Returns:
`pd.Series`: The same record, but with all space separated values converted to arrays
"""
for key in space_sep_fields:
try:
split_list = record[key].split()
if key == "primary_decade":
record[key] = [int(x) for x in split_list]
else:
record[key] = split_list
except AttributeError:
record[key] = []
return record
def main():
# Creates an argument parser that accepts a csv file as input
parser = argparse.ArgumentParser()
parser.add_argument(
"csv_filepath",
help="The path to the CSV file to be converted to json",
)
parser.add_argument(
"type", help="The type of data to be converted", choices=["archive", "stickers"]
)
# Reads the csv file and converts it to a dataframe
args = parser.parse_args()
path = args.csv_filepath
datatype = args.type
df = pd.read_csv(path)
print(f"Successfully read CSV file at {path}")
# Converts all space separated values to arrays
if datatype == "archive":
space_sep_fields = [
"craft_discipline_category",
"craft_discipline",
"primary_decade",
"images",
]
elif datatype == "stickers":
space_sep_fields = []
arrayified_df = df.apply(lambda record: arrayify(
record, space_sep_fields), axis=1)
current_time = datetime.datetime.now().strftime("%Y-%m-%d-%H-%M-%S")
# Saves the dataframe to a json file
new_name_info = f"{datatype}-{current_time}.json"
new_path_info = os.path.join("scripts", "data", "tmp", new_name_info)
# Create the directories if it doesn't exist
os.makedirs(os.path.dirname(new_path_info), exist_ok=True)
arrayified_df.to_json(
new_path_info,
orient="records",
force_ascii=False,
indent=2,
)
print(f"Successfully converted CSV to JSON file at {new_path_info}")
if datatype == "stickers":
print(
f"""
To automatically upload the updated sticker information to the CDDL database,
run the following command:
$ node scripts/upload.js --overwrite scripts/data/tmp/{new_name_info} stickers
This OVERWRITES the existing sticker using the new info
from the CSV.
"""
)
print("SUCCESS")
return
print(
"Detected archive information; creating archive info image metadata fields to update..."
)
# Load the new archive JSON
with open(new_path_info, "r", encoding="utf-8") as f:
archive_info = json.load(f)
image_meta_updates = []
for archive in archive_info:
image_ids = archive["images"]
# Add the image to the archive response and create an image metadata object
if len(image_ids) == 0:
# No images for this archive
continue
# We only have the primary image data for now, so we only update the
# primary image's metadata (the first image)
primary_image_id = image_ids[0]
# Create the updated image meta object
craft_type: list = archive["craft_discipline"][:]
craft_other: str = archive["craft_discipline_other"]
if craft_other is not None:
try:
craft_type.remove("other")
except ValueError:
pass
craft_type.append(craft_other)
# Add the new image metadata (field, value) pairs to be updated
image_meta_updates.append(
{
"img_id": primary_image_id,
"response_id": archive["ID"],
"is_thumbnail": archive["thumb_img_id"] == primary_image_id,
"craft_category": archive["craft_discipline_category"],
"craft_type": craft_type,
"keywords": list(
set(craft_type + archive["craft_discipline_category"])
),
# Geolocation
"location.geo.lat": archive["primary_location.geo.lat"],
"location.geo.lng": archive["primary_location.geo.lng"],
# Address
"location.address.content": archive["primary_location.address.content"],
"location.address.content_ar": archive[
"primary_location.address.content_ar"
],
"location.address.content_orig": archive[
"primary_location.address.content_orig"
],
"location.address.content_orig_lang": archive[
"primary_location.address.content_orig_lang"
],
# Administrative regions
"location.adm4": archive["primary_location.adm4"],
"location.adm3": archive["primary_location.adm3"],
"location.adm2": archive["primary_location.adm2"],
"location.adm1": archive["primary_location.adm1"],
"year_taken": archive["primary_year"],
"decade_taken": archive["primary_decade"],
"historic_map": archive["primary_historic_map"],
"src": f"https://cddl-beirut.herokuapp.com/api/images/{primary_image_id}.jpg",
}
)
# Saves the image metadata (updates) to a json file
new_name_meta = f"archive-updated-image-meta-{current_time}.json"
new_path_meta = os.path.join("scripts", "data", "tmp", new_name_meta)
os.makedirs(os.path.dirname(new_path_meta), exist_ok=True)
with open(new_path_meta, "w", encoding="utf-8") as f:
json.dump(image_meta_updates, f, indent=2, ensure_ascii=False)
print(
f"Successfully wrote archive updated image metadata JSON file at {new_path_meta}"
)
print(
"--------------------------------------------------------------------------------"
)
print(
f"""
To automatically upload the updated archive information and image
metadata to the CDDL API, run the following commands:
$ node scripts/upload.js --overwrite scripts/data/tmp/{new_name_info} archive
This OVERWRITES the existing archive information using the new info
from the CSV.
$ node scripts/upload.js --update scripts/data/tmp/{new_name_meta} image-meta
This UPDATES the existing image metadata using the new info from the
CSV. Does not overwrite the image meta object since the other fields
can only be found by looking at the original responses from the Kobo
survey.
"""
)
print("SUCCESS!")
return
if __name__ == "__main__":
main()