Millefeuille revised this gist . Go to revision
1 file changed, 17 insertions, 14 deletions
ID3Tagger.py
| @@ -3,9 +3,10 @@ import datetime | |||
| 3 | 3 | import json | |
| 4 | 4 | import urllib.request | |
| 5 | 5 | import urllib.parse | |
| 6 | + | import acoustid | |
| 6 | 7 | from mutagen.id3 import ID3NoHeaderError, ID3, TIT2, TALB, TPE2, TPE1, APIC, TRCK, TLEN, TDRC, TYER | |
| 7 | 8 | ||
| 8 | - | base_url = "https://musicbrainz.org/ws/2/recording/" | |
| 9 | + | base_url = "https://musicbrainz.org/ws/2" | |
| 9 | 10 | cover_art_url = f"https://coverartarchive.org/release/" | |
| 10 | 11 | ||
| 11 | 12 | ||
| @@ -14,9 +15,9 @@ def milliseconds_to_duration(milliseconds): | |||
| 14 | 15 | minutes, seconds = divmod(seconds, 60) | |
| 15 | 16 | hours, minutes = divmod(minutes, 60) | |
| 16 | 17 | ||
| 17 | - | duration = "{:02}:{:02}:{:02}.{}".format(hours, minutes, seconds, milliseconds) | |
| 18 | + | music_duration = "{:02}:{:02}:{:02}.{}".format(hours, minutes, seconds, milliseconds) | |
| 18 | 19 | ||
| 19 | - | return duration | |
| 20 | + | return music_duration | |
| 20 | 21 | ||
| 21 | 22 | ||
| 22 | 23 | def get_key(data, key): | |
| @@ -50,7 +51,10 @@ def get_data_from_media(media, tags, title): | |||
| 50 | 51 | if media is None: | |
| 51 | 52 | print("Found no media for this release, no additional data can be recovered") | |
| 52 | 53 | return tags | |
| 53 | - | for track in media['track']: | |
| 54 | + | if 'tracks' not in media or len(media['tracks']) <=0: | |
| 55 | + | print("Found no track for this release, no additional data can be recovered") | |
| 56 | + | return tags | |
| 57 | + | for track in media['tracks']: | |
| 54 | 58 | if track['title'] == title: | |
| 55 | 59 | track_num = f"{track['number']}/{media['track-count']}" | |
| 56 | 60 | print(f"TRCK={track_num}") | |
| @@ -109,32 +113,31 @@ def main(): | |||
| 109 | 113 | prog='metadataFetcher', | |
| 110 | 114 | description='fetches metadata and writes it in a mp3 file' | |
| 111 | 115 | ) | |
| 116 | + | parser.add_argument('api_key') | |
| 112 | 117 | parser.add_argument('filename') | |
| 113 | - | parser.add_argument('title') | |
| 114 | - | parser.add_argument('artist') | |
| 115 | 118 | args = parser.parse_args() | |
| 116 | 119 | ||
| 120 | + | matches = list(acoustid.match(args.api_key, args.filename)) | |
| 121 | + | if len(matches) <= 0: | |
| 122 | + | print("Could not get a match for the fingerprint, exiting...") | |
| 123 | + | exit(1) | |
| 124 | + | ||
| 117 | 125 | try: | |
| 118 | 126 | tags = ID3(args.filename) | |
| 119 | 127 | except ID3NoHeaderError: | |
| 120 | 128 | print("Adding ID3 header...") | |
| 121 | 129 | tags = ID3() | |
| 122 | 130 | ||
| 123 | - | artist = urllib.parse.quote(args.artist) | |
| 124 | - | title = urllib.parse.quote(args.title) | |
| 125 | - | query = f"?query=artist:{artist}%20AND%20recording:{title}&fmt=json" | |
| 126 | - | with urllib.request.urlopen(f"{base_url}{query}") as response: | |
| 131 | + | with urllib.request.urlopen(f"{base_url}/recording/{matches[0][1]}?fmt=json&inc=releases+artists+media+artist-credits") as response: | |
| 127 | 132 | body = response.read() | |
| 128 | - | data = json.loads(body) | |
| 129 | - | ||
| 130 | - | recording = get_recording(data) | |
| 133 | + | recording = json.loads(body) | |
| 131 | 134 | tags = get_data_from_recording(recording, tags) | |
| 132 | 135 | ||
| 133 | 136 | release = get_release(recording) | |
| 134 | 137 | tags = get_data_from_release(release, tags) | |
| 135 | 138 | if release is not None: | |
| 136 | 139 | media = get_media(release) | |
| 137 | - | get_data_from_media(media, tags, title) | |
| 140 | + | get_data_from_media(media, tags, recording['title']) | |
| 138 | 141 | ||
| 139 | 142 | tags.save() | |
| 140 | 143 | ||
Millefeuille revised this gist . Go to revision
1 file changed, 1 deletion
ID3Tagger.py
| @@ -14,7 +14,6 @@ def milliseconds_to_duration(milliseconds): | |||
| 14 | 14 | minutes, seconds = divmod(seconds, 60) | |
| 15 | 15 | hours, minutes = divmod(minutes, 60) | |
| 16 | 16 | ||
| 17 | - | # Create a formatted string | |
| 18 | 17 | duration = "{:02}:{:02}:{:02}.{}".format(hours, minutes, seconds, milliseconds) | |
| 19 | 18 | ||
| 20 | 19 | return duration | |
Millefeuille revised this gist . Go to revision
1 file changed, 144 insertions
ID3Tagger.py(file created)
| @@ -0,0 +1,144 @@ | |||
| 1 | + | import argparse | |
| 2 | + | import datetime | |
| 3 | + | import json | |
| 4 | + | import urllib.request | |
| 5 | + | import urllib.parse | |
| 6 | + | from mutagen.id3 import ID3NoHeaderError, ID3, TIT2, TALB, TPE2, TPE1, APIC, TRCK, TLEN, TDRC, TYER | |
| 7 | + | ||
| 8 | + | base_url = "https://musicbrainz.org/ws/2/recording/" | |
| 9 | + | cover_art_url = f"https://coverartarchive.org/release/" | |
| 10 | + | ||
| 11 | + | ||
| 12 | + | def milliseconds_to_duration(milliseconds): | |
| 13 | + | seconds, milliseconds = divmod(milliseconds, 1000) | |
| 14 | + | minutes, seconds = divmod(seconds, 60) | |
| 15 | + | hours, minutes = divmod(minutes, 60) | |
| 16 | + | ||
| 17 | + | # Create a formatted string | |
| 18 | + | duration = "{:02}:{:02}:{:02}.{}".format(hours, minutes, seconds, milliseconds) | |
| 19 | + | ||
| 20 | + | return duration | |
| 21 | + | ||
| 22 | + | ||
| 23 | + | def get_key(data, key): | |
| 24 | + | if key not in data or len(data[key]) <= 0: | |
| 25 | + | return None | |
| 26 | + | return data[key][0] | |
| 27 | + | ||
| 28 | + | ||
| 29 | + | def get_recording(data): | |
| 30 | + | return get_key(data, 'recordings') | |
| 31 | + | ||
| 32 | + | ||
| 33 | + | def get_release(recording): | |
| 34 | + | return get_key(recording, 'releases') | |
| 35 | + | ||
| 36 | + | ||
| 37 | + | def get_media(release): | |
| 38 | + | return get_key(release, 'media') | |
| 39 | + | ||
| 40 | + | ||
| 41 | + | def concat_artists(data): | |
| 42 | + | if 'artist-credit' not in data or len(data['artist-credit']) <= 0: | |
| 43 | + | return None | |
| 44 | + | artists = [] | |
| 45 | + | for artist in data['artist-credit']: | |
| 46 | + | artists.append(artist['name']) | |
| 47 | + | return ",".join(artists) | |
| 48 | + | ||
| 49 | + | ||
| 50 | + | def get_data_from_media(media, tags, title): | |
| 51 | + | if media is None: | |
| 52 | + | print("Found no media for this release, no additional data can be recovered") | |
| 53 | + | return tags | |
| 54 | + | for track in media['track']: | |
| 55 | + | if track['title'] == title: | |
| 56 | + | track_num = f"{track['number']}/{media['track-count']}" | |
| 57 | + | print(f"TRCK={track_num}") | |
| 58 | + | tags["TRCK"] = TRCK(encoding=3, text=track_num) | |
| 59 | + | return tags | |
| 60 | + | ||
| 61 | + | ||
| 62 | + | def get_data_from_release(release, tags): | |
| 63 | + | if release is None: | |
| 64 | + | print("Found no release for this recording, no additional data can be recovered") | |
| 65 | + | return tags | |
| 66 | + | print(f"TALB={release['title']}") | |
| 67 | + | tags["TALB"] = TALB(encoding=3, text=release['title']) | |
| 68 | + | release_artists = concat_artists(release) | |
| 69 | + | if release_artists is None: | |
| 70 | + | print("Found no artist for this release") | |
| 71 | + | else: | |
| 72 | + | print(f"TPE2={release_artists}") | |
| 73 | + | tags["TPE2"] = TPE2(encoding=3, text=release_artists) | |
| 74 | + | try: | |
| 75 | + | with urllib.request.urlopen(f"{cover_art_url}{release['id']}/front") as cover_art_response: | |
| 76 | + | print(f"APIC={cover_art_url}{release['id']}/front") | |
| 77 | + | tags["APIC"] = APIC(3, 'image/jpeg', 3, 'Front cover', cover_art_response.read()) | |
| 78 | + | except: | |
| 79 | + | print("Found no cover art for this release") | |
| 80 | + | return tags | |
| 81 | + | ||
| 82 | + | ||
| 83 | + | def get_data_from_recording(recording, tags): | |
| 84 | + | if recording is None: | |
| 85 | + | print("Got no match for this recording, exiting...") | |
| 86 | + | exit(1) | |
| 87 | + | print(f"TIT2={recording['title']}") | |
| 88 | + | tags["TIT2"] = TIT2(encoding=3, text=recording['title']) | |
| 89 | + | if 'length' in recording: | |
| 90 | + | print(f"TLEN={milliseconds_to_duration(recording['length'])}") | |
| 91 | + | tags["TLEN"] = TLEN(encoding=3, text=milliseconds_to_duration(recording['length'])) | |
| 92 | + | track_artists = concat_artists(recording) | |
| 93 | + | if track_artists is None: | |
| 94 | + | print("Found no artist for this track") | |
| 95 | + | else: | |
| 96 | + | print(f"TPE1={track_artists}") | |
| 97 | + | tags["TPE1"] = TPE1(encoding=3, text=track_artists) | |
| 98 | + | if 'first-release-date' not in recording: | |
| 99 | + | print("Found no release date for this recording") | |
| 100 | + | return tags | |
| 101 | + | release_date: datetime.date = datetime.date.fromisoformat(recording['first-release-date']) | |
| 102 | + | print(f"TDRC/TYER={release_date.year}") | |
| 103 | + | tags["TYER"] = TYER(encoding=3, text=f"{release_date.year}") | |
| 104 | + | tags["TDRC"] = TDRC(encoding=3, text=f"{release_date.year}") | |
| 105 | + | return tags | |
| 106 | + | ||
| 107 | + | ||
| 108 | + | def main(): | |
| 109 | + | parser = argparse.ArgumentParser( | |
| 110 | + | prog='metadataFetcher', | |
| 111 | + | description='fetches metadata and writes it in a mp3 file' | |
| 112 | + | ) | |
| 113 | + | parser.add_argument('filename') | |
| 114 | + | parser.add_argument('title') | |
| 115 | + | parser.add_argument('artist') | |
| 116 | + | args = parser.parse_args() | |
| 117 | + | ||
| 118 | + | try: | |
| 119 | + | tags = ID3(args.filename) | |
| 120 | + | except ID3NoHeaderError: | |
| 121 | + | print("Adding ID3 header...") | |
| 122 | + | tags = ID3() | |
| 123 | + | ||
| 124 | + | artist = urllib.parse.quote(args.artist) | |
| 125 | + | title = urllib.parse.quote(args.title) | |
| 126 | + | query = f"?query=artist:{artist}%20AND%20recording:{title}&fmt=json" | |
| 127 | + | with urllib.request.urlopen(f"{base_url}{query}") as response: | |
| 128 | + | body = response.read() | |
| 129 | + | data = json.loads(body) | |
| 130 | + | ||
| 131 | + | recording = get_recording(data) | |
| 132 | + | tags = get_data_from_recording(recording, tags) | |
| 133 | + | ||
| 134 | + | release = get_release(recording) | |
| 135 | + | tags = get_data_from_release(release, tags) | |
| 136 | + | if release is not None: | |
| 137 | + | media = get_media(release) | |
| 138 | + | get_data_from_media(media, tags, title) | |
| 139 | + | ||
| 140 | + | tags.save() | |
| 141 | + | ||
| 142 | + | ||
| 143 | + | if __name__ == '__main__': | |
| 144 | + | main() | |