diff --git a/puller_fits.py b/puller_fits.py index 24b2d0c..b04b8aa 100644 --- a/puller_fits.py +++ b/puller_fits.py @@ -15,55 +15,55 @@ directory_url = r"https://data.ngdc.noaa.gov/platforms/solar-space-observing-sat stored_images_dir = r"..\Data" # directory_url = r"https://data.ngdc.noaa.gov/platforms/solar-space-observing-satellites/goes/goes16/l2/data/suvi-l2-ci094/" # stored_images_dir = r"Z:\NOAA GOES Data\Data\goes16\l2\suvi-l2-ci094" -ignore_folder_names = ["l1b", "goes17", "2017", "2018", "2019", "2020", "2021", "2022"] +ignore_folder_names = ["l1b", "goes17", "2017", "2018", "2019", "2020"] file_database_path = r"..\file_database.json" -fetch_interval = 60*60 -nworkers = 16 -randomize_order = True +fetch_interval = 0 # 60*60 +nworkers = 8 +randomize_order = False def recursive_find_links(url, attempt_count = 3, randomize_order = True): links_regex_pattern = r'(?<=.*\d{4}-\d{2}-\d{2} \d{2}:\d{2})' # Find href links that do not contain question marks or whitespace times_regex_pattern = r'(?<=)(\d{4}-\d{2}-\d{2} \d{2}:\d{2})(?= )' # Find timestamps in UTC in the format YYYY-MM-DD HH:mm - with urllib.request.urlopen(url) as response: - attempts = 0 - html_content = None - while attempts < attempt_count: - try: + attempts = 0 + html_content = None + while attempts < attempt_count: + try: + with urllib.request.urlopen(url) as response: html_content = response.read().decode('utf-8') break - except Exception as e: - print(f"Exception while fetching links: {e}") - time.sleep(1 + random.random()) - attempts += 1 - if (attempt_count > 1) and (attempts == attempt_count): - print(f'\nAfter {attempt_count} retries, could not fetch: {url}') - return + except Exception as e: + print(f"Exception while fetching links: {e}") + time.sleep(1 + random.random()) + attempts += 1 + if (attempt_count > 1) and (attempts == attempt_count): + print(f'\nAfter {attempt_count} retries, could not fetch: {url}') + return - links = re.findall(links_regex_pattern, html_content) - times = re.findall(times_regex_pattern, html_content) - for i in range(len(times)): - dt = datetime.datetime.strptime(times[i].strip(), "%Y-%m-%d %H:%M") - dt.replace(tzinfo=datetime.timezone.utc) - times[i] = time.mktime(dt.timetuple()) - - if len(links) - 1 == len(times): - links = links[1:] - elif len(links) == len(times): - pass - else: - raise(ValueError) + links = re.findall(links_regex_pattern, html_content) + times = re.findall(times_regex_pattern, html_content) + for i in range(len(times)): + dt = datetime.datetime.strptime(times[i].strip(), "%Y-%m-%d %H:%M") + dt.replace(tzinfo=datetime.timezone.utc) + times[i] = time.mktime(dt.timetuple()) + + if len(links) - 1 == len(times): + links = links[1:] + elif len(links) == len(times): + pass + else: + raise(ValueError) - results = list(zip(links, times)) - if randomize_order: - random.shuffle(results) - for _link, _time in results: - if _link.endswith("/"): - if _link.split(r"/")[-2] in ignore_folder_names: - continue - else: - yield from recursive_find_links(urllib.parse.urljoin(url,_link), randomize_order) + results = list(zip(links, times)) + if randomize_order: + random.shuffle(results) + for _link, _time in results: + if _link.endswith("/"): + if _link.split(r"/")[-2] in ignore_folder_names: + continue else: - yield urllib.parse.urljoin(url,_link), _time + yield from recursive_find_links(urllib.parse.urljoin(url,_link), attempt_count, randomize_order) + else: + yield urllib.parse.urljoin(url,_link), _time # Fetch image from url and store it to path, retrying on failure @@ -120,7 +120,7 @@ if __name__ == "__main__": already_had_image_count = 0 failed_image_count = 0 urllen = len(directory_url) - for l, t in tqdm.tqdm(recursive_find_links(directory_url, randomize_order), desc="Downloading files"): + for l, t in tqdm.tqdm(recursive_find_links(directory_url, randomize_order = randomize_order), desc="Downloading files"): # Collect completed jobs and record completion status while True: try: @@ -152,7 +152,14 @@ if __name__ == "__main__": with open(file_database_path, 'w') as f: f.write(json.dumps(file_info_cache)) print(f"Downloaded {fetched_image_count} | Already had {already_had_image_count} | Failed {failed_image_count}") - time.sleep(fetch_interval) + if fetch_interval > 0: + print(f"Run complete!, sleeping for {fetch_interval} seconds.") + with open(file_database_path, 'w') as f: + f.write(json.dumps(file_info_cache)) + time.sleep(fetch_interval) + else: + print("Run complete!, exiting...") + break except KeyboardInterrupt: print("Saving file database and shutting down.")