Skip to content
Draft
Show file tree
Hide file tree
Changes from all commits
Commits
Show all changes
20 commits
Select commit Hold shift + click to select a range
06fd10a
Add musicbrainzngs as dependency
veeara282 Mar 23, 2026
7b40924
Begin implementing MusicBrainz crawler
veeara282 Mar 23, 2026
d443c0d
Read package ver. from pyproject.toml for user-agent string
veeara282 Mar 24, 2026
f63cdae
ignore data/musicbrainz folder (contains large API results)
veeara282 Mar 24, 2026
1e230e2
Move helper functions to utils.musicbrainz_helpers module
veeara282 Mar 24, 2026
5761899
Create separate pipeline for OST data from MusicBrainz
veeara282 Mar 24, 2026
3283c76
Add function to extract mbz result sets into normalized DataFrames
veeara282 Mar 24, 2026
231f070
housekeeping: reset_index after deduplicating DataFrame
veeara282 Mar 24, 2026
37fc7e4
Combine datasets
veeara282 Mar 24, 2026
befde8c
Commit AGENTS.md
veeara282 Mar 24, 2026
63fa63c
Apply filter to keep only main series titles
veeara282 Mar 24, 2026
a6c1b05
Fix bug & expand filtering to match Nintendo Music releases
veeara282 Mar 24, 2026
80c92a2
Create subsets for official English and Japanese releases
veeara282 Mar 24, 2026
9f02565
Refactor: Only request official releases to reduce network and comput…
veeara282 Mar 25, 2026
87ed61f
Accommodate responses from XML or JSON APIs
veeara282 Apr 1, 2026
70065ef
Prevent pipelines container from running automatically on docker comp…
veeara282 Apr 1, 2026
44babc1
Add volume for pipeline local data storage
veeara282 Apr 1, 2026
0e2278a
Write intermediate datasets to local storage
veeara282 Apr 1, 2026
8d1638f
Logging timestamps
veeara282 Apr 1, 2026
6a822e5
Move explanatory comments
veeara282 Apr 6, 2026
File filter

Filter by extension

Filter by extension


Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
10 changes: 9 additions & 1 deletion .env-example
Original file line number Diff line number Diff line change
Expand Up @@ -28,4 +28,12 @@ RUSTFS_CONSOLE_PORT=0000
RUSTFS_ACCESS_KEY="********************" #generated using openssl rand -base64 20 | clipcopy
RUSTFS_SECRET_KEY="********************************" #generated using openssl rand -base64 32 | clipcopy

API_PORT=0000
API_PORT=0000

# Contact info (email or URL) for MusicBrainz crawler user agent, as required by MusicBrainz API usage guidelines.
# See https://musicbrainz.org/doc/MusicBrainz_API/Rate_Limiting for details.
PIPELINES_CRAWLER_CONTACT = "dev-email@example.org"

# Local directory to store data generated by pipelines.
# This directory is mounted as a volume in the pipelines container.
PIPELINES_DATA_DIR=/data
10 changes: 10 additions & 0 deletions AGENTS.md
Original file line number Diff line number Diff line change
@@ -0,0 +1,10 @@
- Python code should conform to the Black formatter's style guide. Some of the most
important style rules are:
- String literals are generally delimited by "double quotes". Strings that contain
quotation marks may be formatted using single quotes, e.g. 'quoted "text"'.
- Lines should generally be at most 88 characters long, although comments can go
slightly over.
Generated code need not perfectly adhere to the Black style rules. It may be helpful
to run Black (`uv run black .` from the `pipelines` directory) after generating code.
- Comments should concisely summarize the parts of the code that follow them and should
not over-explain implementation details.
7 changes: 7 additions & 0 deletions docker-compose.yml
Original file line number Diff line number Diff line change
Expand Up @@ -40,10 +40,17 @@ services:
AWS_ENDPOINT_URL_S3: ${S3_ENDPOINT}
AWS_DEFAULT_REGION: us-east-1
S3_BUCKET: ${S3_BUCKET}
PIPELINES_CRAWLER_CONTACT: ${PIPELINES_CRAWLER_CONTACT}
volumes:
- pipeline-data:${PIPELINES_DATA_DIR}
depends_on:
db:
condition: service_healthy
profiles:
# Prevent service from starting automatically when running `docker compose up`
- no-autostart

volumes:
pgdata:
objdata:
pipeline-data:
1 change: 1 addition & 0 deletions pipelines/data/.gitignore
Original file line number Diff line number Diff line change
@@ -1,4 +1,5 @@
# API results should not be checked into Git
*.json
musicbrainz/
soundexchange/
spotify/
114 changes: 114 additions & 0 deletions pipelines/pokemon/get_ost_tracklist_and_work_data.py
Original file line number Diff line number Diff line change
@@ -0,0 +1,114 @@
from datetime import datetime
import logging
from os import mkdir

import musicbrainzngs as mbz
import pandas as pd

from utils.local_io import DATA_DIR
import utils.musicbrainz_helpers as mbz_helpers

logger = logging.getLogger(__name__)

# Uncomment this line to show debug logs from musicbrainzngs
logging.basicConfig(
level=logging.DEBUG,
format="%(asctime)s - %(name)s - %(levelname)s - %(message)s",
)


def get_ost_releases():
# Get MusicBrainz release entries linked to either The Pokémon Company as label
# or Game Freak as artist.
# We filter for official release entries on MusicBrainz as they have more complete
# metadata (including ISRCs, underlying musical works, and composer credits), and
# exclude "bootleg" releases such as gamerips.
# We also query release groups as they may be helpful for matching English and
# Japanese releases of the same game soundtrack.

# The Pokémon Company (Japan)
tpc_label = "e19f9e2b-4dd5-4f52-9e3c-46b678986698"

tpc_releases = mbz.browse_releases(
label=tpc_label,
includes=["release-groups"],
release_status="official",
limit=100, # the maximum size of a single API results page
) #: ReleaseList

# Game Freak
game_freak_artist = "88c8f9c2-763b-45c9-863f-da3c7c6c8fd1"

game_freak_releases = mbz.browse_releases(
artist=game_freak_artist,
includes=["release-groups"],
release_status="official",
limit=100,
) #: ReleaseList

tpc_data_normalized = mbz_helpers.to_dataframes(tpc_releases)
game_freak_data_normalized = mbz_helpers.to_dataframes(game_freak_releases)

combined_data = mbz_helpers.combine_datasets(
tpc_data_normalized, game_freak_data_normalized
)
return combined_data


def filter_main_series(df: pd.DataFrame) -> pd.DataFrame:
# Filter for main series Pokémon OSTs based on title and disambiguation fields.
# This function can take either the releases or the release_groups DataFrame as
# input, since they both have the same "title" field.
return df[
# Filter titles matching "Pokémon" and...
df["title"].str.contains("Pok[eé]mon|ポケモン|ポケットモンスター")
& (
# either "Super Music [Collection/Complete]" (or similar) in the title,
# or "Nintendo Music" in the disambiguation
df["title"].str.contains("Super Music|スーパー|ミュージック")
| (
~df["disambiguation"].isna()
& df["disambiguation"].str.contains("Nintendo Music")
)
)
].reset_index(drop=True)


def main():
mbz_helpers.setup()

# Construct the directory path for this pipeline run using the current timestamp
# in ISO 8601 format with milliseconds precision, e.g. "2026-03-05T12:34:56.789".
dt = datetime.now().isoformat(timespec="milliseconds")
data_dir_for_run = DATA_DIR / "pokemon_ost_data" / dt

# First download all official release entries linked to The Pokémon Company or
# Game Freak as label or artist, then filter for Pokémon main series relevance.
# This function returns a dictionary of DataFrames, which are written to the same
# directory using the mbz_helpers.write_dataset() function.
combined_releases = get_ost_releases()
mbz_helpers.write_dataset(combined_releases, data_dir_for_run / "all_releases")

# Prepare directory for filtered datasets.
# Each filtering step returns a single DataFrame, so the standard to_parquet()
# method is used here instead of mbz_helpers.write_dataset().
main_series_releases_dir = data_dir_for_run / "main_series_releases"
main_series_releases_dir.mkdir(parents=True, exist_ok=True)

main_series_releases = filter_main_series(combined_releases["releases"])
main_series_releases.to_parquet(
main_series_releases_dir / "all.parquet", index=False
)

# Separate out English and Japanese releases based on country code ("XW", "JP").
# The worldwide release entries on MusicBrainz (country code "XW") include English
# track titles, which are useful for matching OST musical works to fan-made covers.
english_releases = main_series_releases[main_series_releases["country"] == "XW"]
japanese_releases = main_series_releases[main_series_releases["country"] == "JP"]

english_releases.to_parquet(main_series_releases_dir / "en.parquet", index=False)
japanese_releases.to_parquet(main_series_releases_dir / "ja.parquet", index=False)


if __name__ == "__main__":
main()
32 changes: 32 additions & 0 deletions pipelines/pokemon/musicbrainz_crawler.py
Original file line number Diff line number Diff line change
@@ -0,0 +1,32 @@
import logging

import musicbrainzngs as mbz

import utils.musicbrainz_helpers as mbz_helpers

logger = logging.getLogger(__name__)

# Uncomment this line to show debug logs from musicbrainzngs
# logging.basicConfig(level=logging.DEBUG)


def main():
mbz_helpers.setup()

# TODO: Retrieve data on Pokémon-related releases, recordings, artists, and works
# using the MusicBrainz API.
# Start with seed data (known artists, soundtrack albums, etc.) and crawl adjacent
# entities in the MusicBrainz graph, prioritizing relevant nodes using best-first search.
# Store retrieved data in a structured format (preferably Parquet) for further processing and analysis.
# MusicBrainz MBIDs are UUIDs and can be stored as `pyarrow.UuidType` or equivalent
# in Parquet files for efficient joining.

known_artists = [
"42c981ec-aa76-43ee-bd60-dd8b2a3d8857", # "Pokémon" artist profile
"cbcafc6f-aa61-400b-9902-00fae1af4a91", # GlitchxCity
"b8e8e0aa-0945-47c0-9c27-1263975e63ae", # Cindery
]


if __name__ == "__main__":
main()
1 change: 1 addition & 0 deletions pipelines/pyproject.toml
Original file line number Diff line number Diff line change
Expand Up @@ -6,6 +6,7 @@ readme = "README.md"
requires-python = ">=3.13"
dependencies = [
"boto3>=1.42.56",
"musicbrainzngs>=0.7.1",
"pandas>=3.0.0",
"pyarrow>=23.0.1",
"python-dateutil>=2.9.0.post0",
Expand Down
5 changes: 5 additions & 0 deletions pipelines/utils/local_io.py
Original file line number Diff line number Diff line change
@@ -0,0 +1,5 @@
import os
from pathlib import Path

"""Root directory for local data storage. This can be overridden by setting the PIPELINES_DATA_DIR environment variable."""
DATA_DIR = Path(os.getenv("PIPELINES_DATA_DIR", "/data"))
136 changes: 136 additions & 0 deletions pipelines/utils/musicbrainz_helpers.py
Original file line number Diff line number Diff line change
@@ -0,0 +1,136 @@
import os
from pathlib import Path
import tomllib

import musicbrainzngs as mbz
import pandas as pd

type ResultSet = dict[str, list[dict]]


def get_package_version():
# Read the package version from pyproject.toml to include in the MusicBrainz API
# user agent string (in accordance with the DRY principle).
path = Path("pyproject.toml")
with path.open("rb") as f:
data = tomllib.load(f)
version = data.get("project", {}).get("version")
return version


def setup():
# Set up MusicBrainz API client with appropriate user agent and rate limiting.
# See https://musicbrainz.org/doc/MusicBrainz_API/Rate_Limiting for details.
# The queries made by this crawler do not require authentication.
package_version = get_package_version()
useragent_contact = os.getenv(
"PIPELINES_CRAWLER_CONTACT", default="https://github.com/veeara282/vgmcharts"
)
mbz.set_useragent("VGMCharts", package_version, contact=useragent_contact)

# Set default rate limit (1 request per second)
mbz.set_rate_limit(limit_or_interval=1.0, new_requests=1)

# Use JSON API
# mbz.set_format("json")


def to_dataframes(result_set: ResultSet) -> dict[str, pd.DataFrame]:
# Extracts all list-based result sets into separate DataFrames, and normalizes
# nested objects into separate DataFrames with foreign key relationships.
# Currently, this function only handles the "releases" result set and its nested
# "release-group" objects.
normalized_dfs: dict[str, pd.DataFrame] = {}

# Look for either "releases" or "release-list" key - "releases" is returned by the
# JSON parser whereas "release-list" is returned by the XML parser.
if "releases" in result_set:
release_list = result_set["releases"]
elif "release-list" in result_set:
release_list = result_set["release-list"]

releases_df = pd.json_normalize(release_list, max_level=0)

# Rename primary key now
releases_df = releases_df.rename(columns={"id": "release_id"})

# Check if release groups were returned in this result set
if "release-group" in releases_df.columns:
has_release_group = releases_df["release-group"].apply(
lambda x: isinstance(x, dict)
)

# Extract release groups into a separate DataFrame
release_groups_df = (
pd.json_normalize(releases_df.loc[has_release_group, "release-group"])
.drop_duplicates(subset=["id"], ignore_index=True)
.rename(columns={"id": "release_group_id"})
.rename(columns=lambda x: x.replace("-", "_"))
)
normalized_dfs["release_groups"] = release_groups_df

# Replace nested object with FK
releases_df["release_group_id"] = releases_df["release-group"].apply(
lambda x: x.get("id") if isinstance(x, dict) else pd.NA
)

releases_df = releases_df.drop(columns=["release-group"])

releases_df = releases_df.rename(columns=lambda x: x.replace("-", "_"))
normalized_dfs["releases"] = releases_df

return normalized_dfs


def write_dataset(dataset: dict[str, pd.DataFrame], output_dir: Path):
# Writes each DataFrame in the dataset to a separate CSV file in the output
# directory.
output_dir.mkdir(parents=True, exist_ok=True)
for name, df in dataset.items():
df.to_parquet(output_dir / f"{name}.parquet", index=False)


def combine_datasets(*dfs: list[dict[str, pd.DataFrame]]) -> dict[str, pd.DataFrame]:
# Combines multiple normalized datasets into a single dataset, merging DataFrames
# with the same name and concatenating rows.
if len(dfs) == 1:
return dfs[0]
if len(dfs) == 0:
raise ValueError("At least one dataset expected")

combined_dfs: dict[str, pd.DataFrame] = {}
grouped_dfs: dict[str, list[pd.DataFrame]] = {}

# Group DataFrames by name across all datasets
for dataset in dfs:
for name, df in dataset.items():
# Append DataFrame to the list if it exists, otherwise append to empty list
grouped_dfs.setdefault(name, []).append(df)

for name, frames in grouped_dfs.items():
# If there is only one DataFrame for this name, no need to do anything
if len(frames) == 1:
combined_dfs[name] = frames[0]
continue

# First combine all rows into a new DataFrame
merged_df = pd.concat(frames, ignore_index=True, sort=False)

# DataFrame may contain multiple "id" columns (primary and foreign keys), so we
# deduplicate based on all of them. There should always be at least one "id"
# column, but we implement a fallback just in case.
keys = [
column
for column in merged_df.columns
if column == "id" or column.endswith("_id")
]

if keys:
merged_df = merged_df.drop_duplicates(subset=keys, ignore_index=True)
else:
# Fallback behavior (if no "id" columns found): just drop duplicate rows
merged_df = merged_df.drop_duplicates(ignore_index=True)

combined_dfs[name] = merged_df

return combined_dfs
11 changes: 11 additions & 0 deletions pipelines/uv.lock

Some generated files are not rendered by default. Learn more about how customized files appear on GitHub.