Import optimizations (not sure if this is helpful). Some fixes and comments.

This commit is contained in:
Johannes Findeisen 2022-11-21 22:54:32 +01:00
commit 418337004a

107
fgg
View file

@ -30,31 +30,27 @@
# https://github.com/alexis-mignon/python-flickr-api/wiki/API-reference # https://github.com/alexis-mignon/python-flickr-api/wiki/API-reference
# This will become the next generation of fgg! # This will become the next generation of fgg!
import crc32c
import flickr_api as flickr import flickr_api as flickr
import hashlib
import json
import pathlib
import os
import requests
import sys
import time
# Set the title for Linux and macOS. from crc32c import crc32c
sys.stdout.write("\x1b]2;fgg\x07") from hashlib import sha1, sha256, sha512, sha3_256, sha3_512
# Set the title for Windows. from json import dump, load
os.system("fgg") from os import environ
from os.path import exists
from pathlib import Path
from requests import get
from time import mktime, strptime
__version__ = "0.1.2" __version__ = "0.1.3"
flickr_api_key = os.environ['FLICKR_API_KEY'] flickr_api_key = environ['FLICKR_API_KEY']
flickr_api_secret = os.environ['FLICKR_API_SECRET'] flickr_api_secret = environ['FLICKR_API_SECRET']
flickr.set_keys(api_key=flickr_api_key, flickr.set_keys(api_key=flickr_api_key,
api_secret=flickr_api_secret) api_secret=flickr_api_secret)
auth_handler = flickr.auth.AuthHandler( auth_handler = flickr.auth.AuthHandler(
access_token_key=os.environ['FLICKR_ACCESS_TOKEN'], access_token_key=environ['FLICKR_ACCESS_TOKEN'],
access_token_secret=os.environ['FLICKR_ACCESS_SECRET']) access_token_secret=environ['FLICKR_ACCESS_SECRET'])
flickr.set_auth_handler(auth_handler) flickr.set_auth_handler(auth_handler)
user = flickr.test.login() user = flickr.test.login()
@ -68,11 +64,11 @@ timezone = 'UTC+01:00'
def hash_file(filename, algorithm='sha1'): def hash_file(filename, algorithm='sha1'):
if algorithm == 'sha1': if algorithm == 'sha1':
h = hashlib.sha1() h = sha1()
if algorithm == 'sha256': if algorithm == 'sha256':
h = hashlib.sha256() h = sha256()
if algorithm == 'sha512': if algorithm == 'sha512':
h = hashlib.sha512() h = sha512()
with open(filename, 'rb') as file: with open(filename, 'rb') as file:
chunk = 0 chunk = 0
@ -105,23 +101,29 @@ print()
print('Page count: ' + str(page_count)) print('Page count: ' + str(page_count))
print('Photo count: ' + str(photo_count)) print('Photo count: ' + str(photo_count))
i = photo_count photo_number = photo_count
photo_process_number = 1
overall_tags = [] overall_tags = []
photo_page = 1 photo_page = 1
while photo_page <= page_count: while photo_page <= page_count:
print("Page: " + str(photo_page)) print("Page: " + str(photo_page))
# TODO: Make this cacheable. Use JSON for that and merge results of all pages to one dict.
# E.g.: flickr_photo_list should be a JSON object not a flickr_api object.
# Then there is no need to loop pages around all code below but only for generating the dict.
# It will make fgg to use only the cache without any Flickr API calls except the count pages
# code above. But I will find to a full cache based solution some day... ;)
flickr_photo_list = user.getPhotos(sort='date-taken-desc', flickr_photo_list = user.getPhotos(sort='date-taken-desc',
per_page=photos_per_page, per_page=photos_per_page,
page=photo_page) page=photo_page)
for flickr_photo in flickr_photo_list: for flickr_photo in flickr_photo_list:
print(str(i) + "/" + str(photo_count)) print(str(photo_process_number) + "/" + str(photo_count))
photo = {} photo = {}
if os.path.exists('cache/' + flickr_photo.id + '/meta.json'): if exists('cache/' + flickr_photo.id + '/meta.json'):
with open('cache/' + flickr_photo.id + '/meta.json', 'rb') as f: with open('cache/' + flickr_photo.id + '/meta.json', 'rb') as f:
print('Loading photo from cache file: cache/' + flickr_photo.id + '/meta.json') print('Loading photo from cache file: cache/' + flickr_photo.id + '/meta.json')
photo = json.load(f) photo = load(f)
f.close() f.close()
else: else:
print('Creating photo cache file: cache/' + flickr_photo.id + '/meta.json') print('Creating photo cache file: cache/' + flickr_photo.id + '/meta.json')
@ -129,8 +131,8 @@ while photo_page <= page_count:
photo['date_posted_timestamp'] = int(flickr_photo_info['posted']) photo['date_posted_timestamp'] = int(flickr_photo_info['posted'])
photo['date_taken'] = flickr_photo_info['taken'] photo['date_taken'] = flickr_photo_info['taken']
photo['date_taken_timestamp'] = int(time.mktime( photo['date_taken_timestamp'] = int(mktime(
time.strptime(flickr_photo_info['taken'], '%Y-%m-%d %H:%M:%S'))) strptime(flickr_photo_info['taken'], '%Y-%m-%d %H:%M:%S')))
photo['date_updated_timestamp'] = int(flickr_photo_info['lastupdate']) photo['date_updated_timestamp'] = int(flickr_photo_info['lastupdate'])
photo['date_uploaded_timestamp'] = int(flickr_photo_info['dateuploaded']) photo['date_uploaded_timestamp'] = int(flickr_photo_info['dateuploaded'])
@ -139,26 +141,28 @@ while photo_page <= page_count:
# START ################################################################################ # START ################################################################################
# NEVER CHANGE THE FOLLOWING LINES (UP TO "# END") UNLESS YOU REALLY NEED OR WANT TO # NEVER CHANGE THE FOLLOWING LINES (UP TO "# END") UNLESS YOU REALLY NEED OR WANT TO
# CHANGE THE PHOTO ID!!! ALL LINKS TO THE PHOTO PAGE WILL BECOME BROKEN!!! - hanez # CHANGE THE PHOTO ID!!! ALL LINKS TO THE PHOTO PAGE WILL BECOME BROKEN!!! - hanez
digest1 = hashlib.sha256(str.encode(str(flickr_photo.id) + digest1 = sha256(str.encode(str(flickr_photo.id) +
str(flickr_photo_info['taken']))).hexdigest() str(flickr_photo_info['taken']))).hexdigest()
digest2 = hashlib.sha512(str.encode(str(flickr_photo.id) + digest2 = sha512(str.encode(str(flickr_photo.id) +
str(flickr_photo_info['taken']))).hexdigest() str(flickr_photo_info['taken']))).hexdigest()
digest3 = hashlib.sha3_256(str.encode(str(flickr_photo.id) + digest3 = sha3_256(str.encode(str(flickr_photo.id) +
str(flickr_photo_info['taken']))).hexdigest() str(flickr_photo_info['taken']))).hexdigest()
digest4 = hashlib.sha3_512(str.encode(str(flickr_photo.id) + digest4 = sha3_512(str.encode(str(flickr_photo.id) +
str(flickr_photo_info['taken']))).hexdigest() str(flickr_photo_info['taken']))).hexdigest()
digest = str(str(digest1) + str(digest2) + str(digest3) + str(digest4)) digest = str(str(digest1) + str(digest2) + str(digest3) + str(digest4))
digest = str(hex(crc32c.crc32c(bytes(digest, 'ascii')))).replace('0x', '') digest = str(hex(crc32c.crc32c(bytes(digest, 'ascii')))).replace('0x', '')
photo['digest'] = digest photo['digest'] = digest
# END ################################################################################## # END ##################################################################################
photo['fgg_id'] = str(photo['date_taken_timestamp']) + '-' + str(i).rjust(7, '0') photo['fgg_id'] = str(photo['date_taken_timestamp']) + '-' + \
str(photo_number).rjust(7, '0')
photo['flickr_id'] = flickr_photo.id photo['flickr_id'] = flickr_photo.id
photo['photo_number'] = i photo['photo_number'] = photo_number
flickr_photo_tags = [] flickr_photo_tags = []
for flickr_photo_tag in flickr_photo_info['tags']: for flickr_photo_tag in flickr_photo_info['tags']:
@ -174,17 +178,17 @@ while photo_page <= page_count:
photo['timezone'] = timezone photo['timezone'] = timezone
photo['title'] = flickr_photo.title photo['title'] = flickr_photo.title
if os.path.exists('cache/' + flickr_photo.id + '/sizes.json'): if exists('cache/' + flickr_photo.id + '/sizes.json'):
with open('cache/' + flickr_photo.id + '/sizes.json', 'rb') as f: with open('cache/' + flickr_photo.id + '/sizes.json', 'rb') as f:
print('Loading sizes from cache file: cache/' + flickr_photo.id + '/sizes.json') print('Loading sizes from cache file: cache/' + flickr_photo.id + '/sizes.json')
flickr_photo_sizes = json.load(f) flickr_photo_sizes = load(f)
f.close() f.close()
else: else:
pathlib.Path('cache/' + flickr_photo.id).mkdir(parents=True, exist_ok=True) Path('cache/' + flickr_photo.id).mkdir(parents=True, exist_ok=True)
with open('cache/' + flickr_photo.id + '/sizes.json', 'w') as f: with open('cache/' + flickr_photo.id + '/sizes.json', 'w') as f:
print('Creating sizes cache file: cache/' + flickr_photo.id + '/sizes.json') print('Creating sizes cache file: cache/' + flickr_photo.id + '/sizes.json')
flickr_photo_sizes = flickr_photo.getSizes() flickr_photo_sizes = flickr_photo.getSizes()
json.dump(flickr_photo_sizes, f, ensure_ascii=False, indent=2, sort_keys=True) dump(flickr_photo_sizes, f, ensure_ascii=False, indent=2, sort_keys=True)
f.close() f.close()
photo['sizes'] = {} photo['sizes'] = {}
@ -248,16 +252,16 @@ while photo_page <= page_count:
photo['sizes'][flickr_size]['target_file_name'] photo['sizes'][flickr_size]['target_file_name']
# Download photo file if it not exists. # Download photo file if it not exists.
if not os.path.exists(photo['sizes'][flickr_size]['target_file']): if not exists(photo['sizes'][flickr_size]['target_file']):
url = flickr_photo_sizes[flickr_photo_size]['source'] url = flickr_photo_sizes[flickr_photo_size]['source']
req = requests.get(url) request = get(url)
print('Downloading: ' + url) print('Downloading file: ' + url)
pathlib.Path(photo['sizes'][flickr_size]['target_file'].rsplit('/', 1)[0]) \ Path(photo['sizes'][flickr_size]['target_file'].rsplit('/', 1)[0]) \
.mkdir(parents=True, exist_ok=True) .mkdir(parents=True, exist_ok=True)
print('Writing file to: ' + photo['sizes'][flickr_size]['target_file']) print('Writing file to: ' + photo['sizes'][flickr_size]['target_file'])
with open(photo['sizes'][flickr_size]['target_file'], 'wb') as f: with open(photo['sizes'][flickr_size]['target_file'], 'wb') as f:
f.write(req.content) f.write(request.content)
f.close() f.close()
print("Download Completed!") print("Download Completed!")
@ -277,17 +281,18 @@ while photo_page <= page_count:
photo['sizes'][flickr_size].pop('source') photo['sizes'][flickr_size].pop('source')
photo['sizes'][flickr_size].pop('url') photo['sizes'][flickr_size].pop('url')
pathlib.Path('cache/' + flickr_photo.id).mkdir(parents=True, exist_ok=True) Path('cache/' + flickr_photo.id).mkdir(parents=True, exist_ok=True)
with open('cache/' + flickr_photo.id + '/meta.json', 'w') as f: with open('cache/' + flickr_photo.id + '/meta.json', 'w') as f:
json.dump(photo, f, ensure_ascii=False, indent=2, sort_keys=True) dump(photo, f, ensure_ascii=False, indent=2, sort_keys=True)
f.close() f.close()
with open(photo['target_path'] + '/meta.json', 'w') as f: with open(photo['target_path'] + '/meta.json', 'w') as f:
json.dump(photo, f, ensure_ascii=False, indent=2, sort_keys=True) dump(photo, f, ensure_ascii=False, indent=2, sort_keys=True)
f.close() f.close()
i -= 1 photo_number -= 1
if i == photo_count - debug_max_photos: photo_process_number += 1
if photo_number == photo_count - debug_max_photos:
exit(0) exit(0)
photo_page += 1 photo_page += 1
@ -295,5 +300,5 @@ while photo_page <= page_count:
photos = {} photos = {}
photos['tags'] = overall_tags photos['tags'] = overall_tags
with open('photos/meta.json', 'w') as f: with open('photos/meta.json', 'w') as f:
json.dump(photos, f, ensure_ascii=False, indent=2, sort_keys=True) dump(photos, f, ensure_ascii=False, indent=2, sort_keys=True)
f.close() f.close()