2026-08-28 01:39:18 -04:00
import os
import urllib . request
import urllib . parse
import re
import time
import random
import datetime
import json
from threading import Thread
import queue
import math
from functools import partial
from queue import Empty
import tqdm
directory_url = r " https://data.ngdc.noaa.gov/platforms/solar-space-observing-satellites/goes/ "
stored_images_dir = os . path . abspath ( os . path . join ( " .. " , " Data " ) )
# directory_url = r"https://data.ngdc.noaa.gov/platforms/solar-space-observing-satellites/goes/goes16/l2/data/ephe-l2-orb1m/2017/12/"
# stored_images_dir = r"Z:\NOAA GOES Data\Data\goes16/l2/data/ephe-l2-orb1m/2017/12/"
ignore_folder_names = [ " Parent Directory " , " l1b " , " goes17 " , " 2017 " , " 2018 " , " 2019 " , " 2020 " , " 2021 " , " 2022 " , " 2023 " ]
file_database_path = os . path . abspath ( os . path . join ( " .. " , " file_database.json " ) )
fetch_interval = 0 # 60*60
nfetchworkers = 2 # Be nice to the servers, this value is how many threads will be asking for links and file info at the same time
ndownloadworkers = 3 # Be nice to the servers, this value is how many threads will be downloading files at the same time
randomize_order = False
links_regex_pattern = r ' (?<=<a href= " )([^ ?:]*)(?= " >.* \ d {4} - \ d {2} - \ d {2} \ d {2} : \ d {2} ) ' # Find href links that do not contain question marks or whitespace
times_regex_pattern = r ' (?<=<td align= " right " >)( \ d {4} - \ d {2} - \ d {2} \ d {2} : \ d {2} )(?= ) ' # Find timestamps in UTC in the format YYYY-MM-DD HH:mm
sizes_regex_pattern = r ' (?<= \ d {4} - \ d {2} - \ d {2} \ d {2} : \ d {2} < \ /td><td align= " right " >)([ . \ d] { 1,3})([KMG]|- |0 | )(?=< \ /td>) ' # Find file size markers in one of many possible formats
links_matcher = re . compile ( links_regex_pattern )
times_matcher = re . compile ( times_regex_pattern )
sizes_matcher = re . compile ( sizes_regex_pattern )
def find_links_worker ( query_work_queue , query_result_queue , attempt_count = 3 , randomize_order = True ) :
while True :
job = query_work_queue . get ( )
if job == None :
return
url = job
attempts = 0
html_content = None
while attempts < attempt_count :
try :
with urllib . request . urlopen ( url ) as response :
html_content = response . read ( ) . decode ( ' utf-8 ' )
break
except Exception as e :
tqdm . tqdm . write ( f " Exception while fetching links: { e } " )
time . sleep ( 1 + random . random ( ) )
attempts + = 1
if ( attempt_count > 1 ) and ( attempts == attempt_count ) :
tqdm . tqdm . write ( f ' \n After { attempt_count } retries, could not fetch: { url } ' )
if html_content is None : # We failed to fetch the link for some reason, continue to the next job
continue
else :
links = links_matcher . findall ( html_content )
times = times_matcher . findall ( html_content )
sizes = sizes_matcher . findall ( html_content )
for i in range ( len ( times ) ) :
dt = datetime . datetime . strptime ( times [ i ] . strip ( ) , " % Y- % m- %d % H: % M " )
dt . replace ( tzinfo = datetime . timezone . utc )
times [ i ] = time . mktime ( dt . timetuple ( ) )
for i in range ( len ( sizes ) ) :
match sizes [ i ] [ 1 ] . strip ( ) :
case ' - ' :
sizes [ i ] = 0
case ' K ' :
sizes [ i ] = int ( float ( sizes [ i ] [ 0 ] . strip ( ) ) * 1024 )
case ' M ' :
sizes [ i ] = int ( float ( sizes [ i ] [ 0 ] . strip ( ) ) * 1024 * 1024 )
case ' G ' :
sizes [ i ] = int ( float ( sizes [ i ] [ 0 ] . strip ( ) ) * 1024 * 1024 * 1024 )
case " 0 " :
sizes [ i ] = 0
case " " :
sizes [ i ] = int ( float ( sizes [ i ] [ 0 ] . strip ( ) ) )
case _ :
raise ( ValueError ( f " Unexpected symbol while parsing links page: { _ } " ) )
if ( len ( links ) != len ( times ) ) or ( len ( times ) != len ( sizes ) ) :
raise ( ValueError ( " Links parsing error, mismatched numbers of links, times, or sizes! " ) )
results = list ( zip ( links , times , sizes ) )
if randomize_order :
random . shuffle ( results )
for _link , _time , _size in results :
if _link . endswith ( " / " ) :
if _link . split ( r " / " ) [ - 2 ] in ignore_folder_names :
continue
else :
# print(f"Queueing folder: {urllib.parse.urljoin(url,_link)}")
query_work_queue . put ( urllib . parse . urljoin ( url , _link ) )
else :
# print(f"Queueing file: {urllib.parse.urljoin(url,_link)}")
query_result_queue . put ( ( urllib . parse . urljoin ( url , _link ) , _time , _size ) )
# Fetch file from url and store it to path, retrying on failure
def file_download_worker ( download_work_queue , download_result_queue , attempt_count = 2 ) :
while True :
job = download_work_queue . get ( )
if job == None :
return
url , path , t , s = job
# print(f"Next job: {url}, {path}, {t}, {s}")
attempts = 0
while attempts < attempt_count :
try :
starttime = time . time ( )
req = urllib . request . Request ( url , data = None )
image_data = urllib . request . urlopen ( req , timeout = 10.0 ) . read ( )
endtime = time . time ( )
tqdm . tqdm . write ( f " Downloaded { url } in { endtime - starttime : 0.2f } s | { len ( image_data ) / 1024 / 1024 / ( endtime - starttime ) : 0.2f } MB/s " )
if math . isclose ( len ( image_data ) , s , rel_tol = 0.05 ) :
os . makedirs ( os . path . split ( path ) [ 0 ] , exist_ok = True )
open ( path , ' wb ' ) . write ( image_data )
download_result_queue . put ( ( True , url , t ) )
break
else :
raise ValueError ( " Downloaded file is the wrong size! " )
except Exception as e :
if hasattr ( e , " code " ) and e . code == 404 : # This is expected if the file has been removed from the site (at least for swpc.noaa.gov)
attempts + = attempt_count
elif ( 0 < attempts ) and ( attempts < attempt_count ) :
# tqdm.tqdm.write(f"\nA problem occurred on file: {url} | {e}")
time . sleep ( 1 + random . random ( ) )
attempts + = 1
if ( attempt_count > 1 ) and ( attempts == attempt_count ) :
tqdm . tqdm . write ( f ' \n After { attempt_count } retries, could not fetch: { url } ' )
tqdm . tqdm . write ( f " Exception: { e } " )
download_result_queue . put ( ( False , url , t ) )
break
if __name__ == " __main__ " :
file_info_cache = { }
try :
print ( f " Attempting to load file records from cache: { file_database_path } " )
with open ( file_database_path , ' r ' ) as f :
file_info_cache = json . loads ( f . read ( ) )
print ( f " File records loaded from cache: { len ( file_info_cache ) } records found. " )
except Exception as e :
print ( f " Load failed, starting with empty cache " )
file_info_cache = { }
query_work_queue = queue . Queue ( )
query_result_queue = queue . Queue ( )
download_work_queue = queue . Queue ( maxsize = ndownloadworkers )
download_result_queue = queue . Queue ( maxsize = ndownloadworkers )
workers = [ ]
for _ in range ( nfetchworkers ) :
t = Thread ( target = find_links_worker , args = ( query_work_queue , query_result_queue , 3 , randomize_order ) , daemon = True )
t . start ( )
workers . append ( t )
for _ in range ( ndownloadworkers ) :
t = Thread ( target = file_download_worker , args = ( download_work_queue , download_result_queue ) , daemon = True )
t . start ( )
workers . append ( t )
try :
while True :
fetched_image_count = 0
already_had_image_count = 0
failed_image_count = 0
urllen = len ( directory_url )
query_work_queue . put ( directory_url )
for l , t , s in tqdm . tqdm ( iter ( partial ( query_result_queue . get , timeout = 30.0 ) , None ) , desc = " Downloading files " ) :
# Collect completed jobs and record completion status
while True :
try :
r_success , r_url , r_t = download_result_queue . get_nowait ( )
if r_success :
fetched_image_count + = 1
file_info_cache [ r_url ] = r_t
else :
failed_image_count + = 1
except queue . Empty :
break
file_portion_of_link = l [ urllen : ]
filepath = os . path . join ( stored_images_dir , file_portion_of_link . replace ( " / " , os . sep ) )
# If we dont have the file or the file at the link is newer than the one we previously fetched
if ( not ( l in file_info_cache ) ) or t > file_info_cache [ l ] :
if filepath . endswith ( " .fits " ) :
filepath2 = filepath . split ( " .fits " ) [ 0 ] + " _f.fits " # also look for the filtered version of the file
filepath3 = filepath . split ( " .fits " ) [ 0 ] + " _e.fits " # also look for the error version of the file
if os . path . exists ( filepath2 ) or os . path . exists ( filepath3 ) : # If the unfiltered filename exists, that case will be handled in the alternative code path in if os.path.exists(filepath):
if l in file_info_cache : # If we have record of this file, it must be out of date, delete it and download the new version.
try :
os . remove ( filepath2 )
os . remove ( filepath3 )
except :
pass
else : # If we have no record of this file, update the file info cache and don't redownload
file_info_cache [ l ] = t
already_had_image_count + = 1
continue
if os . path . exists ( filepath ) :
if l in file_info_cache : # If we have record of this file, it must be out of date, delete it and download the new version.
if os . path . exists ( filepath ) :
os . remove ( filepath )
else : # If we have no record of this file, update the file info cache and don't redownload if fsize is right
fsize = os . path . getsize ( filepath )
if math . isclose ( fsize , s , rel_tol = 0.05 ) :
file_info_cache [ l ] = t
already_had_image_count + = 1
continue
else :
tqdm . tqdm . write ( f ' Found a mismatched size on file: { filepath } Redownloading! ' )
download_work_queue . put ( ( l , filepath , t , s ) )
else :
# We have a download record, confirm the file actually exists on disk
if os . path . exists ( filepath ) :
already_had_image_count + = 1
elif filepath . endswith ( " .fits " ) :
filepath2 = filepath . split ( " .fits " ) [ 0 ] + " _f.fits " # also look for the filtered version of the file
filepath3 = filepath . split ( " .fits " ) [ 0 ] + " _e.fits " # also look for the error version of the file
if os . path . exists ( filepath2 ) or os . path . exists ( filepath3 ) :
already_had_image_count + = 1
else : # We could not find the file on disk, queue for redownload
download_work_queue . put ( ( l , filepath , t ) )
with open ( file_database_path , ' w ' ) as f :
f . write ( json . dumps ( file_info_cache ) )
print ( f " Downloaded { fetched_image_count } | Already had { already_had_image_count } | Failed { failed_image_count } " )
if fetch_interval > 0 :
print ( f " Run complete!, sleeping for { fetch_interval } seconds. " )
with open ( file_database_path , ' w ' ) as f :
f . write ( json . dumps ( file_info_cache ) )
time . sleep ( fetch_interval )
else :
print ( " Run complete!, exiting... " )
break
except KeyboardInterrupt :
print ( " Saving file database and shutting down. " )
except Empty :
print ( " Work Complete, Shutting down... " )
except Exception as e :
print ( f " Unhandled Exception during run: { e } " )
print ( " Shutting down " )
for _ in range ( nfetchworkers ) :
try :
query_work_queue . put ( None )
except :
break
time . sleep ( 1 )
print ( f " Downloaded { fetched_image_count } | Already had { already_had_image_count } | Failed { failed_image_count } " )
while len ( download_work_queue . queue ) > 0 :
print ( f " Waiting for { len ( download_work_queue . queue ) } downloads in queue... " )
time . sleep ( 1 )
for _ in range ( ndownloadworkers ) :
try :
download_work_queue . put ( None , timeout = 5.0 )
except :
break
with open ( file_database_path , ' w ' ) as f :
f . write ( json . dumps ( file_info_cache ) )
print ( " Waiting for workers to shutdown... " )
for w in workers :
2024-07-05 10:24:41 -04:00
w . join ( 5.0 )