Repository navigation
Expand file tree
/
Copy pathebird.py
More file actions
773 lines (639 loc) · 32.7 KB
/
Copy pathebird.py
File metadata and controls
773 lines (639 loc) · 32.7 KB
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
129
130
131
132
133
134
135
136
137
138
139
140
141
142
143
144
145
146
147
148
149
150
151
152
153
154
155
156
157
158
159
160
161
162
163
164
165
166
167
168
169
170
171
172
173
174
175
176
177
178
179
180
181
182
183
184
185
186
187
188
189
190
191
192
193
194
195
196
197
198
199
200
201
202
203
204
205
206
207
208
209
210
211
212
213
214
215
216
217
218
219
220
221
222
223
224
225
226
227
228
229
230
231
232
233
234
235
236
237
238
239
240
241
242
243
244
245
246
247
248
249
250
251
252
253
254
255
256
257
258
259
260
261
262
263
264
265
266
267
268
269
270
271
272
273
274
275
276
277
278
279
280
281
282
283
284
285
286
287
288
289
290
291
292
293
294
295
296
297
298
299
300
301
302
303
304
305
306
307
308
309
310
311
312
313
314
315
316
317
318
319
320
321
322
323
324
325
326
327
328
329
330
331
332
333
334
335
336
337
338
339
340
341
342
343
344
345
346
347
348
349
350
351
352
353
354
355
356
357
358
359
360
361
362
363
364
365
366
367
368
369
370
371
372
373
374
375
376
377
378
379
380
381
382
383
384
385
386
387
388
389
390
391
392
393
394
395
396
397
398
399
400
401
402
403
404
405
406
407
408
409
410
411
412
413
414
415
416
417
418
419
420
421
422
423
424
425
426
427
428
429
430
431
432
433
434
435
436
437
438
439
440
441
442
443
444
445
446
447
448
449
450
451
452
453
454
455
456
457
458
459
460
461
462
463
464
465
466
467
468
469
470
471
472
473
474
475
476
477
478
479
480
481
482
483
484
485
486
487
488
489
490
491
492
493
494
495
496
497
498
499
500
501
502
503
504
505
506
507
508
509
510
511
512
513
514
515
516
517
518
519
520
521
522
523
524
525
526
527
528
529
530
531
532
533
534
535
536
537
538
539
540
541
542
543
544
545
546
547
548
549
550
551
552
553
554
555
556
557
558
559
560
561
562
563
564
565
566
567
568
569
570
571
572
573
574
575
576
577
578
579
580
581
582
583
584
585
586
587
588
589
590
591
592
593
594
595
596
597
598
599
600
601
602
603
604
605
606
607
608
609
610
611
612
613
614
615
616
617
618
619
620
621
622
623
624
625
626
627
628
629
630
631
632
633
634
635
636
637
638
639
640
641
642
643
644
645
646
647
648
649
650
651
652
653
654
655
656
657
658
659
660
661
662
663
664
665
666
667
668
669
670
671
672
673
674
675
676
677
678
679
680
681
682
683
684
685
686
687
688
689
690
691
692
693
694
695
696
697
698
699
700
701
702
703
704
705
706
707
708
709
710
711
712
713
714
715
716
717
718
719
720
721
722
723
724
725
726
727
728
729
730
731
732
733
734
735
736
737
738
739
740
741
742
743
744
745
746
747
748
749
750
751
752
753
754
755
756
757
758
759
760
761
762
763
764
765
766
767
768
769
770
771
772
773
import collections
import json
import json.decoder
import logging
from typing import Any
from urllib.parse import urlparse
import requests
from bs4 import BeautifulSoup
import cache
from queryGoogle import query_google_images_api
# Note: need this hack because BeautifulSoup doesn't work with Python 3.10+
# Discussion at
# https://stackoverflow.com/questions/69515086/error-attributeerror-collections-has-no-attribute-callable-using-beautifu
collections.Callable = collections.abc.Callable
logger = logging.getLogger()
class EBird:
def __init__(self):
"""
Initializes the logger and loads in the key data at startup so that requests are fast
"""
self.__get_groups_dictionary()
def __get_species_code(self, species_name):
"""
Looks up and returns the species ebird code needed for scraping ebird site. An example is
"ameavo" for the "American Avocet" species. Also called the taxonCode by ebird site.
:param species_name:
:return: species code
"""
# Determine the species code, which is needed for scraping the ebird site
taxonomy_dict = self.__get_taxonomy_dictionary()
unified_species_name = self.__unified_name(species_name)
if unified_species_name not in taxonomy_dict:
logger.warning(f'Species={species_name} not found in ebird.get_species_info()')
return None
species_data = taxonomy_dict[unified_species_name]
return species_data['speciesCode']
def __abbreviate_loc(self, loc_str: str):
"""
Abbreviates loc string via substitution, such as ", United States" => ", USA".
This way the location takes up less valuable space when displayed. The commas
are used so that won't get an inappropriate match for a city name.
List of country codes is at https://www.iban.com/country-codes .
State ones are at
https://www.faa.gov/air_traffic/publications/atpubs/cnt_html/appendix_a.html
:param loc_str: location to be abbreviated
:return: abbreviated location
"""
return (loc_str
.replace(", United States", ", USA")
.replace(", Alabama", ", AL")
.replace(", Alaska", ", AK")
.replace(", Arizona", ", AZ")
.replace(", Arkansas", ", AR")
.replace(", American Samoa", ", AS")
.replace(", California", ", CA")
.replace(", Colorado", ", CO")
.replace(", Connecticut", ", CT")
.replace(", Delaware", ", DE")
.replace(", District of Columbia", ", DC")
.replace(", Florida", ", FL")
.replace(", Georgia", ", GA")
.replace(", Guam", ", GU")
.replace(", Hawaii", ", HI")
.replace(", Idaho", ", ID")
.replace(", Illinois", ", IL")
.replace(", Indiana", ", IN")
.replace(", Iowa", ", IA")
.replace(", Kansas", ", KS")
.replace(", Kentucky", ", KY")
.replace(", Louisiana", ", LA")
.replace(", Maine", ", ME")
.replace(", Maryland", ", MD")
.replace(", Massachusetts", ", MA")
.replace(", Michigan", ", MI")
.replace(", Minnesota", ", MN")
.replace(", Mississippi", ", MS")
.replace(", Missouri", ", MO")
.replace(", Montana", ", MT")
.replace(", Nebraska", ", NE")
.replace(", Nevada", ", NV")
.replace(", New Hampshire", ", NH")
.replace(", New Jersey", ", NJ")
.replace(", New Mexico", ", NM")
.replace(", New York", ", NY")
.replace(", North Carolina", ", NC")
.replace(", North Dakota", ", ND")
.replace(", Northern Mariana Islands", ", MP")
.replace(", Ohio", ", OH")
.replace(", Oklahoma", ", OK")
.replace(", Oregon", ", OR")
.replace(", Pennsylvania", ", PA")
.replace(", Puerto Rico", ", PR")
.replace(", Rhode Island", ", RI")
.replace(", South Carolina", ", SC")
.replace(", South Dakota", ", SD")
.replace(", Tennessee", ", TN")
.replace(", Texas", ", TX")
.replace(", Trust Territories", ", TT")
.replace(", Utah", ", UT")
.replace(", Vermont", ", VT")
.replace(", Virginia", ", VA")
.replace(", Virgin Islands", ", VI")
.replace(", Washington", ", WA")
.replace(", West Virginia", ", WV")
.replace(", Wisconsin", ", WI")
.replace(", Wyoming", ", WY")
.replace(", Canada", ", CAN")
.replace(", Alberta", ", AB")
.replace(", British Columbia", ", BC")
.replace(", Newfoundland", ", NF")
.replace(", Ontario", ", ON")
.replace(", Quebec", ", QC")
.replace(", Brazil", ", BRA")
.replace(", Cayman Islands", ", Cayman Is")
.replace(", China", ", CHN")
.replace(", England", ", GB")
.replace(", Germany", ", DEU")
.replace(", India", ", IND")
.replace(", Israel", ", ISR")
.replace(", Mexico", ", MEX")
.replace(", New Zealand", ", NZL")
.replace(", Russia", ", RUS")
.replace(", Saudi Arabia", ", SAU")
.replace(", Scotland", ", GB-SCT")
.replace(", South Africa", ", S Africa")
.replace(", South Korea", ", KOR")
.replace(", Korea", ", KOR")
.replace(", Thailand", ", THA")
)
def __get_audio_data_list_for_species(self, species_name):
"""
Scrapes ebird site to get info on best audio files for the specified species. Not cached since whoever
calls this method caches it.
:param species_name: Which species want info for
:return: list of objects containing image info for species
"""
logger.info(f'Determining audio data, including, urls for species={species_name}...')
# Determine the species code, which is needed for scraping the ebird site
species_code = self.__get_species_code(species_name)
url = (f'https://media.ebird.org/catalog?taxonCode={species_code}'
f'&mediaType=audio&sort=rating_rank_desc&view=list')
# Make request to the website
logger.info(f'Requesting audio info for species={species_name} from url={url}')
req = requests.get(url)
logger.debug(f'Finished receiving audio info for species {species_name}')
# Parse the returned html
logger.debug(f'Parsing audio info html for species {species_name}...')
soup = BeautifulSoup(req.text, 'html.parser')
audio_info_list = []
# Get the results list
ol = soup.find('ol', class_="ResultsList")
li_elements = ol.find_all('li')
for li in li_elements:
header = li.find('div', class_='ResultsList-header')
# Get the URL of the mp3 file. It is the first immediate child of header that is a <a>
first_a = header.find('a', recursive=False)
full_catalog_number = first_a.text.replace('"', '') # Will be something like "ML483235"
catalog_number = full_catalog_number.replace('ML', '')
audio_url = f'https://cdn.download.ams.birds.cornell.edu/api/v2/asset/{catalog_number}/mp3'
tag_span = header.find('span', class_='ResultsList-label')
tag_label_text = tag_span.text if tag_span is not None else None
tag_content_text = tag_span.find_next_sibling('span').text if tag_span is not None else None
# Get the rating
meta_element = li.find('div', class_='ResultsList-meta')
rating_stars_element = meta_element.find('div', class_='RatingStars')
ratings_text = rating_stars_element.find('span').text
rating = ratings_text.replace('rating', '').strip()
# If poor rating then done since they are in order, as long as get at least two
if len(audio_info_list) > 0 and (not rating.isdigit() or int(rating) < 3):
break
# Usually author name is in a <a> element but sometimes it is in a <span>
user_date_loc_element = meta_element.find('div', class_='userDateLoc')
author_element = user_date_loc_element.find(['a', 'span'])
author = author_element.text
# Usually date is in a <time> element but sometimes it is in a span (if it is unknown for example0
date_element = author_element.find_next(['time', 'span'])
date = date_element.text
# Since sometimes author is also in an span need to use date_element.find_next() to
# dependably get the location span
loc_element = date_element.find_next('span')
loc = self.__abbreviate_loc(loc_element.text)
audio_info_list.append({'catalog': full_catalog_number,
'author': author,
'date': date,
'loc': loc,
'audioUrl': audio_url,
'rating': rating,
'tagLabel': tag_label_text,
'tagContent': tag_content_text})
# Just use 10 best. If used more then would rarely have cache hits. And they
# are supposedly listed in order of ratings. Better to just use the 10 best.
if len(audio_info_list) >= 10:
break
logger.debug(f'Done processing audio data for species {species_name}')
return audio_info_list
def __get_image_data_list_for_species(self, species_name):
"""
Scrapes ebird site to get info on best images for the specified species. Not cached since whoever
calls this method caches it.
:param species_name: Which species want info for
:return: list of objects containing image info for species
"""
logger.info(f'Determining image data, including urls, for species={species_name}...')
# Determine the species code, which is needed for scraping the ebird site
species_code = self.__get_species_code(species_name)
url = (f'https://media.ebird.org/catalog?taxonCode={species_code}'
f'&mediaType=photo&sort=rating_rank_desc&view=list')
# Make request to the website
logger.info(f'Requesting image info for species={species_name} from url={url}')
req = requests.get(url)
logger.debug(f'Finished receiving image info for species {species_name}')
# Parse the returned html
logger.debug(f'Parsing image info html for species {species_name}...')
soup = BeautifulSoup(req.text, 'html.parser')
image_info_list = []
# Get the results list
ol = soup.find('ol', class_="ResultsList")
li_elements = ol.find_all('li')
for li in li_elements:
# Get the catalog info
header = li.find('div', class_='ResultsList-header')
first_a = header.find('a', recursive=False)
full_catalog_number = first_a.text.replace('"', '') # Will be something like "ML483235"
media = li.find('div', class_='ResultsList-media')
image = media.find('img')
image_url = image.attrs['src']
# Get the rating
meta_element = li.find('div', class_='ResultsList-meta')
rating_stars_element = meta_element.find('div', class_='RatingStars')
ratings_text = rating_stars_element.find('span').text
rating = ratings_text.replace('rating', '').strip()
# If poor rating then done since they are in order, as long as get at least two
if len(image_info_list) >= 2 and (not rating.isdigit() or int(rating) < 3):
break
# Usually author name is in a <a> element but sometimes it is in a <span>
user_date_loc_element = meta_element.find('div', class_='userDateLoc')
author_element = user_date_loc_element.find(['a', 'span'])
author = author_element.text
# Usually date is in a <time> element but sometimes it is in a span (if it is unknown for example0
date_element = author_element.find_next(['time', 'span'])
date = date_element.text
# Since sometimes author is also in an span need to use date_element.find_next() to
# dependably get the location span
loc_element = date_element.find_next('span')
loc = self.__abbreviate_loc(loc_element.text)
# Get the tags like 'Behavior'
tags_div = li.find('div', class_='ResultsList-tags')
tags_list = tags_div.find_all('div')
found_tags = []
for div in tags_list:
label = div.find('span', class_='ResultsList-label')
content = label.find_next('span')
found_tags.append({'label': label.text, 'content': content.text})
image_info_list.append({'catalog': full_catalog_number,
'author': author,
'date': date,
'loc': loc,
'imageUrl': image_url,
'rating': rating,
'tags': found_tags})
# Just use 10 best. If used more then would rarely have cache hits. And they
# are supposedly listed in order of ratings. Better to just use the 10 best.
if len(image_info_list) >= 10:
break
logger.debug(f'Done processing image info for species {species_name}')
return image_info_list
def __unified_name(self, species_name):
"""
Turns out ebird is not 100% consistent with their species names. Found at least one case,
"Black-crowned Night-Heron" where the second dash isn't always there. Also found a problem
with "Western/Eastern Cattle Egret". Therefore need to do lookups with a unified name.
:rtype: str
:param species_name:
:return: Modified species name that is easier to match
"""
return (species_name.replace('-', ' ')
.replace('Western/Eastern ', '')
.replace('Western Flycatcher (Cordilleran)', 'Cordilleran Flycatcher')
.lower())
def __get_track_data(self):
"""
Loads in audio "track" data from Macaulay Library (Ithaca) and puts it into
an array. This is a useful way of determining the most relevant bird species
to include. A species will likely have multiple audio tracks. Data in the list
is not sorted in any way. Might not actually use the audio tracks from this list
because can get a greater selection by perusing the ebird web page for a species,
but it is definitely useful for determining list of species to use.
:return: Array of data describing each audio track available
"""
# The webpage where the data is listed
url = 'https://www.macaulaylibrary.org/guide-to-bird-sounds/track-list/'
# Make request to the website
logger.info(f'Getting species data from url={url}')
req = requests.get(url)
# Parse the returned html
soup = BeautifulSoup(req.text, 'html.parser')
# Get the first table
table = soup.find_all('table')[0]
# Process each row, after the first header row, for that table
rows = table.find_all('tr')
rows_data = []
for row in rows[1:]:
row_data = {}
cells = row.find_all('td')
row_data['track'] = cells[0].text.strip()
# For title trim off the end recording number and other cruft
species = cells[1].text.strip()
end = species.find(' 0')
if end == -1:
end = species.find(' 1')
if end == -1:
end = species.find(' 2')
# Determine species name, and replace the weird apostrophe with a regular one
row_data['species'] = species[0:end].replace("’", "'")
row_data['callName'] = species
row_data['scientificName'] = cells[2].text.strip()
row_data['recordist'] = cells[3].text.strip()
row_data['location'] = cells[4].text.strip()
row_data['date'] = cells[5].text.strip()
# If no link indicating catalog number then skip this row
a = cells[6].find('a')
if a is None:
continue
catalog_number = a.text.strip()
row_data['catalogNumber'] = catalog_number
# Store link to audio file
number = catalog_number.lstrip('ML')
row_data['audioUrl'] = f'https://cdn.download.ams.birds.cornell.edu/api/v2/asset/{number}/mp3'
# Useful to include copyright info
row_data['copyright'] = 'Cornell Lab Macaulay Library'
rows_data.append(row_data)
return rows_data
def __get_species_tracks_dictionary(self):
"""
Data dictionary is keyed on species name and value is list of audio tracks for the species.
Raw data & track info is from https://www.macaulaylibrary.org/guide-to-bird-sounds/track-list/
Not cached since the calling methods cache it. But I'm not confident this is truly the best
thing to do.
:return: the data dictionary
"""
# Add each species data to the species dictionary
logger.info("Creating species dictionary...")
species_dict = dict()
for species_data in self.__get_track_data():
list_for_species = species_dict.get(species_data['species'])
if list_for_species is None:
list_for_species = []
species_dict[species_data['species']] = list_for_species
list_for_species.append(species_data)
return species_dict
__taxonomy_dictionary_cache = None
def __get_taxonomy_dictionary(self):
"""
Gets as a dictionary the taxonomy of all bird species (~30k!) from the ebird site. Includes ebird
taxonomy name, which is needed for looking up best images and audio clips on ebird. Caches
the dictionary so don't need to keep hitting the ebird site.
:return: taxonomy of all bird species. A dictionary keyed by unified species name and containing basic
info about the species. unified_species_name = self.__unified_name(species_name)
"""
# Try getting from memory cache first
if self.__taxonomy_dictionary_cache is not None:
logger.info(f'Using taxonomy dictionary from memory cache')
return self.__taxonomy_dictionary_cache
# Try getting from file cache
cache_file_name = "allEbirdSpeciesTaxonomyDictionaryCache.json"
if cache.file_exists(cache_file_name):
logger.info(f'Using taxonomy dictionary from file cache')
json_data = cache.read_from_cache(cache_file_name)
return json.loads(json_data)
# Load in the full taxonomy from ebird site
logger.info(f'Generating taxonomy dictionary because was not cached')
url = "https://api.ebird.org/v2/ref/taxonomy/ebird?fmt=json "
headers = {'x-ebirdapitoken': 'jfekjedvescr'}
response = requests.get(url, headers=headers)
full_taxonomy = json.loads(response.content)
taxonomy_dict: dict[Any, dict[str, Any]] = {}
for full_species in full_taxonomy:
# If species missing any important data then skip it
if ('comName' not in full_species or
'speciesCode' not in full_species or
'sciName' not in full_species or
'familyComName' not in full_species):
continue
species_name = full_species['comName']
unified_species_name = self.__unified_name(species_name)
species = {
"speciesName": species_name,
"speciesCode": full_species['speciesCode'],
"sciName": full_species['sciName'],
"groupName": full_species['familyComName']}
taxonomy_dict[unified_species_name] = species
# Write the full_taxonomy json to cache file in nice format by dumping object into json string
cache.write_to_cache(json.dumps(taxonomy_dict, indent=4), cache_file_name)
# Store in memory cache
self.__taxonomy_dictionary_cache = taxonomy_dict
return taxonomy_dict
__supplemental_species_config_cache = None
def __supplemental_species_config(self):
"""
Reads in supplemental species config file supplementalSpeciesConfig.json
:return: data for the supplemental species
"""
if self.__supplemental_species_config_cache is not None:
logger.info(f'Using cached supplemental species config info')
return self.__supplemental_species_config_cache
# Read in supplemental data and convert JSON to a python object
logger.info(f'Generating supplemental species config info')
supplemental_file_name = 'data/supplementalSpeciesConfig.json'
try:
with open(supplemental_file_name, 'rb') as file:
json_data = file.read()
try:
supplemental_species = json.loads(json_data)
except json.decoder.JSONDecodeError as err:
logger.error(f'Error parsing supplementalSpeciesConfig.json {err}')
return {}
except FileNotFoundError:
logger.warning(f'The supplemental file {supplemental_file_name} does not exist')
return {}
# Convert to a dictionary so can look up data by species_name easily
supplemental_species_dict = {}
for species in supplemental_species:
supplemental_species_dict[species['speciesName']] = species
# Cache result
__supplemental_species_config_cache = supplemental_species_dict
return supplemental_species_dict
def __add_species_to_group(self, species_name, group_name, groups):
if group_name not in groups:
species_list_for_group = [species_name]
groups[group_name] = species_list_for_group
else:
species_list_for_group = groups[group_name]
species_list_for_group.append(species_name)
__groups_dictionary_cache = None
def __get_groups_dictionary(self):
"""
Provides the group list for the species specified in the species_list. Each group is a list of
species names within that group. The groups will NOT be in alphabetical order.
:return: dictionary of all groups. Keyed by group name and containing values of list of all
species names for that group
"""
# Return memory cached value if exists
if self.__groups_dictionary_cache is not None:
logger.info(f'Using memory cached groups dictionary')
return self.__groups_dictionary_cache
# Use file cache if it exists
cache_file_name = 'groupsCache.json'
if cache.file_exists(cache_file_name):
logger.info(f'Using file cached groups dictionary')
json_data = cache.read_from_cache(cache_file_name)
return json.loads(json_data)
logger.info("Generating the groups dictionary...")
# The return value. groups is a dictionary keyed on group name and containing list of species names
groups = {}
# So that can limit which species are listed
species_name_list = self.get_species_name_list()
taxonomy = self.__get_taxonomy_dictionary()
# For each species name from __get_species_name_list...
for species_name in species_name_list:
# Get the species info from the taxonomy data
uni_name = self.__unified_name(species_name)
if uni_name not in taxonomy:
# If can't find this species, even using unified name, in the taxonomy, then skip it
logger.warning(f'Could not find species "{species_name}" in taxonomy so skipping it.')
continue
species = taxonomy[uni_name]
# Add the group name to the groups dictionary
group_name = species['groupName']
self.__add_species_to_group(species_name, group_name, groups)
# Read in and add the supplemental data to the groups object
supplemental_species = self.__supplemental_species_config()
for species in supplemental_species.values():
self.__add_species_to_group(species['speciesName'], species['groupName'], groups)
# Write groups to cache
cache.write_to_cache(json.dumps(groups, indent=4), cache_file_name)
# Store in memory cache
self.__groups_dictionary_cache = groups
return groups
def __get_sorted_species_list(self):
"""
Returns list of all species in alphabetical order. For each species there is a list of the
track audio calls, but those particular track calls might not be used. Caches the data.
:return: list of all data, ordered alphabetically by species
"""
# Try getting from cache first
cache_file_name = "speciesTracksListCache.json"
if cache.file_exists(cache_file_name):
logger.info(f'Using file cached species list {cache_file_name}')
json_data = cache.read_from_cache(cache_file_name)
return json.loads(json_data)
logger.info("Generating speciesTrackList...")
species_dictionary = self.__get_species_tracks_dictionary()
keys_list = list(species_dictionary.keys())
keys_list.sort()
all_species_list = []
for species in keys_list:
all_species_list.append(species_dictionary.get(species))
# Write to cache
json_data = json.dumps(all_species_list, indent=4)
cache.write_to_cache(json_data, cache_file_name)
return all_species_list
def get_species_name_list(self):
"""
Returns list of species names in alphabetical order
:return: list of species names
"""
all_data_list = self.__get_sorted_species_list()
species_list = []
for species in all_data_list:
# Each a_species is a list of calls
a_call_for_species = species[0]
species_list.append(a_call_for_species['species'])
return sorted(species_list)
# Memory cache for get_species_list_json()
__species_names_list_cache = None
def get_species_list_json(self):
"""
Returns JSON string of list of species names in alphabetical order.
:return: species list as JSON string
"""
# If in memory cache return it
if self.__species_names_list_cache is not None:
return self.__species_names_list_cache
# Try getting from file cache first
cache_file_name = "speciesNamesListCache.json"
if cache.file_exists(cache_file_name):
logger.info(f'Getting species names list from file cache {cache_file_name}')
return cache.read_from_cache(cache_file_name)
logger.info('Generating species name list json...')
# Get the list of species names
species_list = self.get_species_name_list()
# Convert to JSON
json_data = json.dumps(species_list, indent=4)
# Write to cache
cache.write_to_cache(json_data, cache_file_name)
# Write to memory cache
__species_names_list_cache = json_data
# Return the results in JSON
return json_data
__group_names_list_cache = None
def get_group_list_json(self):
"""
Returns JSON str of list of group names alphabetized
:return: JSON str of group names
"""
# If in memory cache return it
if self.__group_names_list_cache is not None:
logger.info(f'Getting the group name list from memory cache')
return self.__group_names_list_cache
# Try getting from cache first
cache_file_name = "groupNamesListCache.json"
if cache.file_exists(cache_file_name):
logger.info(f'Getting the group name list from file cache {cache_file_name}')
return cache.read_from_cache(cache_file_name)
logger.info('Determining group list json...')
# Get the group info
groups_dict = self.__get_groups_dictionary()
# Put group names into an array
group_names = []
for group_name in groups_dict:
group_names.append(group_name)
# Convert sorted group names to a json str
json_data = json.dumps(sorted(group_names), indent=2)
# Write to cache
cache.write_to_cache(json_data, cache_file_name)
# Write to memory cache
__group_names_list_cache = json_data
# Return the results in JSON
return json_data
def get_species_info(self, species_name):
"""
Returns info for the specified species, including list of image info, list of audio info, and some other
data.
:param species_name:
:return: json str containing info for species
"""
# Return info from cache if available
cache_file_name = 'speciesDataCache.json'
if cache.file_exists(cache_file_name, subdir=species_name):
logger.info(f'Using file cached data for species={species_name} file={cache_file_name}')
return cache.read_from_cache(cache_file_name, subdir=species_name)
logger.info(f'Generating data for species={species_name}...')
# First, see if info for this species is in the supplemental file. This
# way the supplemental file can override what is automatically determined
# using ebird data.
supplemental_species_dict = self.__supplemental_species_config()
if species_name in supplemental_species_dict:
# Use supplemental info
supplemental_species = supplemental_species_dict.get(species_name)
image_search_query = supplemental_species['imageSearchQuery']
image_list = query_google_images_api(image_search_query)
audio_data_list = supplemental_species['audioDataList']
for item in audio_data_list:
if item.get('title') is None:
item['title'] = urlparse(item['audioUrl']).netloc
species_data = {
"speciesName": supplemental_species['speciesName'],
"groupName": supplemental_species['groupName'],
"imageDataList": image_list,
"audioDataList": audio_data_list}
else:
# Get all the ebird info for the species
taxonomy_dict = self.__get_taxonomy_dictionary()
unified_species_name = self.__unified_name(species_name)
if unified_species_name not in taxonomy_dict:
logger.warning(f'Species={species_name} not found in ebird.get_species_info()')
return None
species_data = taxonomy_dict[unified_species_name]
# Add info for images and audio
image_data_list = self.__get_image_data_list_for_species(species_name)
species_data['imageDataList'] = image_data_list
audio_data_list = self.__get_audio_data_list_for_species(species_name)
species_data['audioDataList'] = audio_data_list
# Convert the species data into json
json_data = json.dumps(species_data, indent=2)
# Write to cache
cache.write_to_cache(json_data, cache_file_name, subdir=species_name)
return json_data
def get_species_for_group_json(self, group_name):
"""
Returns json consisting of list of species for the specified group
:param group_name:
:return: list of species for group
"""
groups_dict = self.__get_groups_dictionary()
if group_name not in groups_dict:
return f'Error: group {group_name} does not exist'
species_list = groups_dict[group_name]
json_data = json.dumps(species_list, indent=2)
return json_data
def get_species_by_group_json(self):
"""
Returns json consisting of list of species for each group
:return: list of species by group
"""
groups_dict = self.__get_groups_dictionary()
return json.dumps(groups_dict, indent=2)
# Global instantiation
ebird = EBird()