Converts a Spotify full account export (not the limited export, the full one) to PDS records compatible with Teal.fm's "fm.teal.alpha.feed.play" lexicon
Something went wrong. Try again.
converter.py
· Created 11mo ago ·21 kB · 542 lines
Python
123456789101112131415161718192021222324252627282930313233343536373839404142434445464748495051525354555657585960616263646566676869707172737475767778798081828384858687888990919293949596979899100101102103104105106107108109110111112113114115116117118119120121122123124125126127128129130131132133134135136137138139140141142143144145146147148149150151152153154155156157158159160161162163164165166167168169170171172173174175176177178179180181182183184185186187188189190191192193194195196197198199200201202203204205206207208209210211212213214215216217218219220221222223224225226227228229230231232233234235236237238239240241242243244245246247248249250251252253254255256257258259260261262263264265266267268269270271272273274275276277278279280281282283284285286287288289290291292293294295296297298299300301302303304305306307308309310311312313314315316317318319320321322323324325326327328329330331332333334335336337338339340341342343344345346347348349350351352353354355356357358359360361362363364365366367368369370371372373374375376377378379380381382383384385386387388389390391392393394395396397398399400401402403404405406407408409410411412413414415416417418419420421422423424425426427428429430431432433434435436437438439440441442443444445446447448449450451452453454455456457458459460461462463464465466467468469470471472473474475476477478479480481482483484485486487488489490491492493494495496497498499500501502503504505506507508509510511512513514515516517518519520521522523524525526527528529530531532533534535536537538539540541542543#!/usr/bin/env python3
# This requires some basic knowledge of Python to use. Sorry, this was a quick-and-dirty script, not anything meant for general consumption# Usage:# * Create a venv with the PIP packages "atproto", "unidecode", and "musicbrainzngs" (Google "python venv" if you don't know how to do this)# * Put your Spotify exports in the same folder as this Python file# * Double-check that FILES_TO_CONVERT matches your Spotify export filenames# * Put in your MusicBrainz credentials into MBRNZ_USER/MBRNZ_PASS (yeah I'm not using environment variables, sorry)# * Create an app password for your ATProto/Bluesky account and put it into BSKY_HANDLE and BSKY_PASSWORD, also point BSKY_PDS at your PDS (probably https://bsky.social)# * Run the script (`python ./converter.py`)# * Watch the teal.fm firehose to verify things are going across: https://discord.com/channels/1299158421655912498/1418272538299207891# * (Alternatively, just look at your PDS directly with https://pdsls.dev/)## This process will take a very long time due to Bluesky rate limits.# Bluesky can process 11666 records/day or 1666 records/hour, whichever is lower. Being locked out will lock you out of Bluesky itself too# Expect the process to take multiple days. I wrote the script to be slightly below the rate limits so just be careful how much you like posts etc.## Some limitations:# * We try to fetch the MusicBrainz ID for every song, but it doesn't always work. When it fails, we leave that data unpopulated so it'll backfill# * There are sometimes more subtle failures where the fuzzy matching gets a _slightly_ wrong version (wrong album etc.). This is because Spotify# song/album names do not always match what MusicBrainz has# * You're only "supposed" to scrobble songs which are more than halfway completed - we don't have that info during the scrobble process (probably# could add it, but ehhh) so you sometimes get double-scrobbles when you stopped a track and then resumed listening to it later. (We do handle# skipping songs, so you don't need to worry about accidentally tracking a skip)# * People will get to stare at your library slowly upload to the Teal.FM firehose, and that can be embarassing
# Boring license stuff:# DO WHAT THE FUCK YOU WANT TO PUBLIC LICENSE# Version 2, December 2004## Copyright (C) 2004 Sam Hocevar <sam@hocevar.net>## Everyone is permitted to copy and distribute verbatim or modified# copies of this license document, and changing it is allowed as long# as the name is changed.## DO WHAT THE FUCK YOU WANT TO PUBLIC LICENSE# TERMS AND CONDITIONS FOR COPYING, DISTRIBUTION AND MODIFICATION## 0. You just DO WHAT THE FUCK YOU WANT TO.## (Also I do not take liability if this blows your stuff up, if it fails catastrophically somehow that's on you)
import jsonimport musicbrainzngsimport reimport stringimport timeimport unicodedata
from atproto import Clientfrom datetime import datetime, timedelta, timezonefrom collections import dequefrom unidecode import unidecodefrom zoneinfo import ZoneInfo
# Grab your Spotify exports (the full Spotify data, not the truncated one - there's 2 options, make sure you're grabbing the right one)# Once you have 'em, put them in the same folder as you're putting this Python scriptFILES_TO_CONVERT = [ 'Streaming_History_Audio_2012-2019_0.json', 'Streaming_History_Audio_2019-2022_1.json', 'Streaming_History_Audio_2022-2025_2.json']
# MusicBrainz username and password - yes, I'm hardcoding secrets, just be careful with 'emMBRNZ_USER = YOUR MUSICBRAINZ USERNAME GOES HEREMBRNZ_PASS = YOUR MUSICBRAINZ PASSWORD GOES HEREMBRNZ_CONTACT = YOUR CONTACT INFO SO MUSICBRAINZ CAN YELL AT YOU IF YOU USE ALL THEIR BANDWIDTH
# PDS handle and app password (I am not implementing OAuth for this)BSKY_PDS_URL = YOUR PDS HERE (PROBABLY https://bsky.social)BSKY_HANDLE = YOUR BLUESKY HANDLE WITHOUT THE @BSKY_PASSWORD = YOUR BLUESKY APP PASSWORD
# If you are on a Mushroom PDS (if you don't know what that means, you are on a Mushroom PDS) then Bluesky puts special rate limits on you for hourly/daily usage# Exceeding these limits will lock you out of EVERYTHING on the network, not just Teal but also Bluesky, Tangled, everything ATProtoBSKY_RATE_LIMIT_HOURLY = 1666BSKY_RATE_LIMIT_DAILY = 11666
PROVIDER = "tealfm"MB_CACHE = {}
pds = Client(BSKY_PDS_URL)
punctuation_table = str.maketrans('', '', string.punctuation)
last_cache_miss = 0.0total_songs = 0
musicbrainzngs.auth(MBRNZ_USER, MBRNZ_PASS)musicbrainzngs.set_useragent("Jay's Teal.fm Spotify Importer", "0.1", MBRNZ_CONTACT)
def main(): print("Starting convert\n") start_time = time.perf_counter()
all_songs = []
# Handle the session being interrupted midway through cutoff_ts = datetime.max start_str = None with open(PROVIDER + "_last_ts.txt", "r", encoding="utf-8") as f: start_str = f.readline().strip()
if start_str != None: start_ts = datetime.strptime(start_str, "%Y-%m-%dT%H:%M:%SZ") else: start_ts = datetime.min
# Parse each file for file in FILES_TO_CONVERT: print("Parsing " + file)
with open(file, 'r', encoding='utf-8') as open_file: data = json.load(open_file)
for entry in data: skipped = entry.get('skipped', False) incognito = entry.get('incognito_mode', False) played_seconds = int(entry['ms_played']) / 1000.0
artist = entry['master_metadata_album_artist_name'] track = entry['master_metadata_track_name'] album = entry['master_metadata_album_album_name']
# Skip podcasts/audiobooks if artist is None or track is None: continue if entry.get("episode_name") or entry.get("audiobook_title"): continue
global total_songs total_songs += 1
# Parse timestamp try: ts_utc = datetime.strptime(entry['ts'], "%Y-%m-%dT%H:%M:%SZ") ts = ts_utc.replace(tzinfo=ZoneInfo("UTC")).astimezone(ZoneInfo("America/Los_Angeles")) except Exception: ts_utc = datetime.now(timezone.utc) ts = datetime.now()
if skipped or played_seconds < 30: continue
# Skip anything after our end cutoff if ts_utc > cutoff_ts: continue
# Skip anything we have already imported if ts_utc <= start_ts: continue
# Skip anything we didn't want to log if incognito: continue
formatted = entry['ts'] track_uri = entry['spotify_track_uri'] song_data = [artist, track, album, formatted, artist, str(int(played_seconds)), track_uri] all_songs.append(song_data)
print("Finished parsing songs")
scrobble_to_pds(all_songs)
print("\nConvert finished in " + str(time.perf_counter() - start_time) + " seconds.")
def normalize_key(s): if s is None: return ''
# Decompose Unicode characters (NFKD), remove accents s = unicodedata.normalize('NFKD', s) s = unidecode(s) # Romanize non-Latin scripts s = s.lower().strip()
# Common replacements before cleanup replacements = { r'\boriginal motion picture soundtrack\b': 'ost', r'\boriginal soundtrack\b': 'ost', r'\bsoundtrack\b': 'ost', r'\bvol(\.|ume)?\b': 'vol', r'\bpart\b': 'pt', r'\bparts\b': 'pt', r'\bedition\b': '', r'\bthe\b': '', r'\band\b': '', r'\bep\b': '', r'\bwalt disney records\b': '', r'\blegacy collection\b': '', r'\bgreatest hits\b': '', r'\breissue(d)?\b': '', r'\bre-issue(d)?\b': '', r'\bsong of the\b': '', r'\bost\b': '', r'\bdeluxe\b': '', } for pattern, repl in replacements.items(): s = re.sub(pattern, repl, s, flags=re.IGNORECASE)
# Remove tags like “Remastered”, “Deluxe”, “Expanded”, etc. cleanup_patterns = [ r'\bfrom\b.*$', # remove everything after 'from' r'\(.*\)', r'\[.*(remaster(ed)?|deluxe|expanded|ep|single|greatest hits|anniversary|special edition|bonus tracks?|credits track|version|mix|mono|stereo|reissue(d)?).*?\]', r'[-–:]\s*(remaster(ed)?(\s*\d{4})?|ep|single|deluxe|expanded|anniversary|greatest hits|special edition|ghost note symphonies|country version|bonus tracks?|credits track|version|mix|mono|stereo|reissue(d)?).*$', ] for pattern in cleanup_patterns: s = re.sub(pattern, '', s, flags=re.IGNORECASE)
# Remove trailing artist/cover info s = re.sub(r'\s*-\s*(?:cover|live|remaster|remix|version|edit|single|mono|stereo|mix|karaoke|instrumental|feat\.?|featuring)\b.*$', '', s, flags=re.IGNORECASE) s = re.sub(r'\(.*cover.*?\)', '', s, flags=re.IGNORECASE) # removes "(Pink Floyd cover)" s = re.sub(r'\[.*cover.*?\]', '', s, flags=re.IGNORECASE) # removes "[Pink Floyd cover]" s = re.sub(r'\s*(feat\.?|featuring)\s+.*$', '', s, flags=re.IGNORECASE)
# Standardize quotes/apostrophes s = s.replace("’", "'").replace("‘", "'").replace("“", '"').replace("”", '"') s = ''.join(c for c in s if not unicodedata.combining(c))
# Remove punctuation s = s.translate(punctuation_table)
# Convert Roman numerals I–XX to numbers roman_map = { 'xx': '20', 'xix': '19', 'xviii': '18', 'xvii': '17', 'xvi': '16', 'xv': '15', 'xiv': '14', 'xiii': '13', 'xii': '12', 'xi': '11', 'x': '10', 'ix': '9', 'viii': '8', 'vii': '7', 'vi': '6', 'v': '5', 'iv': '4', 'iii': '3', 'ii': '2', 'i': '1' } for roman, arabic in roman_map.items(): s = re.sub(rf'\b{roman}\b', arabic, s, flags=re.IGNORECASE)
# Convert written numbers one–twenty to digits word_nums = { 'one': '1', 'two': '2', 'three': '3', 'four': '4', 'five': '5', 'six': '6', 'seven': '7', 'eight': '8', 'nine': '9', 'ten': '10', 'eleven': '11', 'twelve': '12', 'thirteen': '13', 'fourteen': '14', 'fifteen': '15', 'sixteen': '16', 'seventeen': '17', 'eighteen': '18', 'nineteen': '19', 'twenty': '20' } for word, num in word_nums.items(): s = re.sub(rf'\b{word}\b', num, s, flags=re.IGNORECASE)
# Strip punctuation again and normalize spaces s = re.sub(r'[^\w\s]', '', s) s = re.sub(r'\s+', '', s)
return s
def scrobble_to_pds(all_songs): # Login using your Bluesky/PDS credentials pds.login(BSKY_HANDLE, BSKY_PASSWORD)
remaining = len(all_songs) request_times = deque() failed_lookups = {}
last_pds_access_time = 0.0
# Favor the daily limit if there's any potential for it being close if len(all_songs) >= BSKY_RATE_LIMIT_DAILY * 0.75: time_between_posts = (60 * 60 * 24) / (BSKY_RATE_LIMIT_DAILY * 0.9) else: # Can skirt a little closer to the hourly limit time_between_posts = (60 * 60) / (BSKY_RATE_LIMIT_HOURLY * 0.95)
print("") print("") print("") print("Rate limit: " + str(time_between_posts))
# Wait a bit in case the job was re-kicked time.sleep(time_between_posts)
print(f"Starting PDS scrobble at {time.time()}") scrobble_begin_time = time.perf_counter()
for entry in all_songs: start_time = time.perf_counter() artist = entry[0] title = entry[1] album = entry[2] timestamp = entry[3] url = entry[6]
mbrn_track = lookup_track(artist, album, title)
if mbrn_track.get("releaseMbId") is None: track_norm = normalize_key(title) album_norm = normalize_key(album) key = f"{normalize_key(artist)}::{album_norm}::{track_norm}" print(f"⚠️ No release found for {key}")
retry_time = 2
time_remaining = remaining * time_between_posts eta_utc = datetime.now(timezone.utc) + timedelta(seconds=time_remaining) eta_local = eta_utc.astimezone() if time_remaining < 60: time_remaining_str = str(time_remaining) + " seconds" else: time_remaining /= 60 if time_remaining < 60: time_remaining_str = str(time_remaining) + " minutes" else: time_remaining /= 60 if time_remaining < 48: time_remaining_str = str(time_remaining) + " hours" else: time_remaining /= 24 time_remaining_str = str(time_remaining) + " days"
print(f"{title} - {artist} ({album}) - listened to at {timestamp}") print("Remaining records to be processed: " + str(remaining) + "; " + str(round((1.0 - (remaining / total_songs)) * 100)) + "% complete") print("⏳ ETA: " + time_remaining_str + " (finishing on " + eta_local.strftime("%A, %Y-%m-%d at %I:%M:%S %p %Z") + ")")
# --- 2. Create the record data --- record = { "$type": "fm.teal.alpha.feed.play", "playedTime": timestamp, "artists": mbrn_track.get("artists", [{"artistName": artist}]), "trackName": mbrn_track.get("trackName", title), "recordingMbId": mbrn_track.get("recordingMbId"), "releaseName": mbrn_track.get("releaseName", album), "releaseMbId": mbrn_track.get("releaseMbId"), "duration": mbrn_track.get("duration"), "submissionClientAgent": "manual/unknown", "musicServiceBaseDomain": "spotify.com", "originUrl": url } if mbrn_track.get("releaseMbId") is None: track_norm = normalize_key(title) album_norm = normalize_key(album) key = f"{normalize_key(artist)}::{album_norm}::{track_norm}" failed_lookups[key] = record
success = False while not success: try: delta_time = time.perf_counter() - last_pds_access_time sleep_time = time_between_posts - delta_time if sleep_time > 0: print("Waiting " + str(sleep_time) + " before trying to contact PDS to stay under Mushroom rate limits") time.sleep(sleep_time) print("") print("Pushing to PDS")
now = time.time() request_times.append(now) last_pds_access_time = time.perf_counter()
# --- 3. Publish the record to your repo --- response = pds.com.atproto.repo.create_record( data={ "repo": pds.me.did, "collection": "fm.teal.alpha.feed.play", "record": record, } ) success = True except Exception as e: print(f"⚠️ Error posting {title}: {e}, waiting {retry_time} seconds") time.sleep(retry_time) retry_time *= 2 if retry_time > 14400: retry_time = 2
print("✅ Record " + str(record) + " created successfully!") print("AT URI:" + response.uri + "; CID: " + response.cid) print("")
# Drop timestamps older than one hour while request_times and request_times[0] < now - 3600: request_times.popleft()
remaining -= 1
with open(PROVIDER + "_last_ts.txt", "w", encoding="utf-8") as f: f.write(f"{entry[3]}")
time_spent = time.perf_counter() - start_time total_time_spent = time.perf_counter() - scrobble_begin_time
# --- 4. Print results --- print("Took " + str(time_spent)) print("Requests so far this hour: " + str(len(request_times)) + " (working for " + str(total_time_spent / 60.0) + " minutes)") print("")
print("") print("")
def get_from_cache(key): """Return cached result if valid, else None.""" return MB_CACHE.get(key)
def set_cache(key, recording, release): """Store result in cache with timestamp."""
if recording != None: metadata = { "trackName": recording["title"], "recordingMbId": recording["id"], "duration": int(int(recording.get("length", 0)) / 1000) if "length" in recording else None, }
if "artist-credit" in recording: metadata["artists"] = []
for credit in recording["artist-credit"]: if "artist" in credit: a = credit["artist"] metadata["artists"].append({"artistName": a["name"], "artistMbId": a["id"]})
metadata["releaseName"] = release.get("title", None) metadata["releaseMbId"] = release.get("id", None) else: print(f"⚠️ Couldn't find recording: " + key) metadata = {}
MB_CACHE[key] = {"data": metadata, "timestamp": time.time()} print("Cached " + key + ": " + str(metadata)) print("New cache size " + str(len(MB_CACHE))) print("") print("")
def match_artist(artist_input, artist_credits): """Check if any artist in the credits matches the input.""" artist_input_norm = normalize_key(artist_input) for credit in artist_credits: if "artist" in credit: mb_artist_record = credit["artist"] artist_key = normalize_key(mb_artist_record["name"]) if artist_key == artist_input_norm: return True
for alias in mb_artist_record.get('alias-list', []): if normalize_key(alias['alias']) == artist_input_norm: return True
print("Not same artist: " + artist_key + " != " + artist_input_norm) return False
def parse_date(date_str): """Parse MusicBrainz release date safely.""" if not date_str: return None try: # handle partial dates like "2001-05" or "2001" parts = date_str.split("-") if len(parts) == 1: return datetime(int(parts[0]), 1, 1) elif len(parts) == 2: return datetime(int(parts[0]), int(parts[1]), 1) else: return datetime(int(parts[0]), int(parts[1]), int(parts[2])) except Exception: return None
def lookup_track(artist_name, album_name, track_name): """ Look up a track on MusicBrainz by artist and title. Returns the best match or None. """ global last_cache_miss
track_norm = normalize_key(track_name) album_norm = normalize_key(album_name) key = f"{normalize_key(artist_name)}::{album_norm}::{track_norm}" cached = get_from_cache(key) if cached: return cached["data"] miss_time = time.perf_counter() delta_time = miss_time - last_cache_miss print("Cache miss for " + key + ", first miss in " + str(delta_time) + " seconds")
sleep_time = 1.0 - delta_time if sleep_time > 0: print("Waiting " + str(sleep_time) + " before trying to fetch track metadata") time.sleep(sleep_time)
try: # Grab several results in case we need to match the album result = musicbrainzngs.search_recordings( artist=artist_name, recording=track_name, limit=10, ) last_cache_miss = time.perf_counter() recordings = result.get("recording-list", [])
if not recordings: set_cache(key, None, None) return get_from_cache(key)["data"]
# Filter by album if provided candidates = [] for rec in recordings: rec_artists = rec.get("artist-credit", []) if not match_artist(artist_name, rec_artists): continue
rec_track_norm = normalize_key(rec.get("title")) if rec_track_norm != track_norm: print("Not same track: " + rec_track_norm + " != " + track_norm) continue
for release in rec.get("release-list", []): rec_album_norm = normalize_key(release.get("title")) status = release.get("status", "") if status != "Official": print("Non-official: " + rec_album_norm + " (" + status +")") continue
release_date = parse_date(release.get("date")) candidates.append((release_date, rec, release)) if rec_album_norm != album_norm: print("Not same album: " + rec_album_norm + " != " + album_norm) continue
print("Matched: " + album_norm)
# Perfect match set_cache(key, rec, release) return get_from_cache(key)["data"] # Couldn't find anything set_cache(key, None, None) return get_from_cache(key)["data"] except musicbrainzngs.WebServiceError as e: print(f"MusicBrainz lookup error: {e}") return {}
main()