Restructure the remaining commands into the house shape
Every command now runs as one orchestrating class (the ctor stores, run() executes, helpers and constants private), main a thin controller; the guards the fixed corpus cannot trigger are dropped, docstrings say each level's own contract once, and the build-db summary reports the songs and the artists alone. Co-Authored-By: Claude Opus 5 <noreply@anthropic.com>
This commit is contained in:
1 parent
443e1ad328
commit
71000892f1
11 files changed
+1919
-1882
No files matched your search
@@ -30,8 +30,9 @@ import time
|
||||
import urllib.parse
|
||||
import urllib.request
|
||||
from collections.abc import Sequence
|
||||
from dataclasses import dataclass
|
||||
from pathlib import Path
|
||||
from typing import Any
|
||||
from typing import Any, ClassVar
|
||||
|
||||
import sqlalchemy as sa
|
||||
from sqlalchemy.orm import Session
|
||||
@@ -45,60 +46,6 @@ from ..models import (
|
||||
)
|
||||
from ..utils import format_duration
|
||||
|
||||
PROVENANCE_FIELDS: Sequence[str] = (
|
||||
"song_id", "source", "method", "acquired_at", "note")
|
||||
"""The header columns of the lyrics provenance CSV file."""
|
||||
USER_AGENT: str = ("pop-fem-audit-tools"
|
||||
" (https://github.com/imacat/pop-fem-audit)")
|
||||
"""The User-Agent header sent on every HTTP request."""
|
||||
TIMEOUT: float = 30.0
|
||||
"""The timeout of an HTTP request, in seconds."""
|
||||
SLEEP_SECONDS: float = 1.0
|
||||
"""The delay between consecutive HTTP requests, in seconds."""
|
||||
|
||||
|
||||
def __build_normalization() -> dict[int, str | None]:
|
||||
"""Build the lyrics normalization translation table.
|
||||
|
||||
:return: The codepoint-to-replacement mapping, a replacement
|
||||
of None meaning removal.
|
||||
"""
|
||||
table: dict[int, str | None] = {}
|
||||
codepoint: int
|
||||
for codepoint in range(0x80, 0xa0):
|
||||
try:
|
||||
table[codepoint] = bytes([codepoint]).decode("cp1252")
|
||||
except UnicodeDecodeError:
|
||||
table[codepoint] = None
|
||||
table[0x0435] = "e"
|
||||
table[0x03cc] = "ó"
|
||||
for codepoint in (0x2005, 0x205f, 0x200a):
|
||||
table[codepoint] = " "
|
||||
for codepoint in (0x200b, 0x200c, 0x200d, 0xfeff):
|
||||
table[codepoint] = None
|
||||
return table
|
||||
|
||||
|
||||
NORMALIZATION: dict[int, str | None] = __build_normalization()
|
||||
"""The codepoint-to-replacement mapping applied to fetched
|
||||
lyrics: cp1252-mojibake restoration for U+0080-U+009F (with the
|
||||
five byte values undefined in cp1252 removed), homoglyph
|
||||
restoration for the Cyrillic "e" and the Greek "o" with tonos,
|
||||
ASCII-space restoration for exotic space variants, and removal
|
||||
of zero-width characters. A replacement of None removes the
|
||||
codepoint."""
|
||||
|
||||
|
||||
def normalize_lyrics(text: str) -> str:
|
||||
"""Restore or remove watermark and mojibake characters.
|
||||
|
||||
:param text: The lyrics text as fetched from an API.
|
||||
:return: The text with the codepoints in
|
||||
:data:`NORMALIZATION` replaced or removed; every other
|
||||
character is unchanged.
|
||||
"""
|
||||
return text.translate(NORMALIZATION)
|
||||
|
||||
|
||||
def parse_args(argv: list[str] | None) -> argparse.Namespace:
|
||||
"""Parse the command-line arguments.
|
||||
@@ -122,6 +69,16 @@ def parse_args(argv: list[str] | None) -> argparse.Namespace:
|
||||
class LyricsFetcher:
|
||||
"""A fetcher of song lyrics from the public lyrics APIs."""
|
||||
|
||||
__USER_AGENT: ClassVar[str] = (
|
||||
"pop-fem-audit-tools"
|
||||
" (https://github.com/imacat/pop-fem-audit)")
|
||||
"""The User-Agent header sent on every HTTP request."""
|
||||
__TIMEOUT: ClassVar[float] = 30.0
|
||||
"""The timeout of an HTTP request, in seconds."""
|
||||
__SLEEP_SECONDS: ClassVar[float] = 1.0
|
||||
"""The delay between consecutive HTTP requests, in
|
||||
seconds."""
|
||||
|
||||
def __init__(self) -> None:
|
||||
"""Construct the fetcher."""
|
||||
self.__sent: int = 0
|
||||
@@ -193,78 +150,216 @@ class LyricsFetcher:
|
||||
network, or decoding error.
|
||||
"""
|
||||
if self.__sent > 0:
|
||||
time.sleep(SLEEP_SECONDS)
|
||||
time.sleep(self.__SLEEP_SECONDS)
|
||||
self.__sent += 1
|
||||
request: urllib.request.Request = urllib.request.Request(
|
||||
url, headers={"User-Agent": USER_AGENT})
|
||||
url, headers={"User-Agent": self.__USER_AGENT})
|
||||
try:
|
||||
with urllib.request.urlopen(
|
||||
request, timeout=TIMEOUT) as response:
|
||||
request, timeout=self.__TIMEOUT) as response:
|
||||
return json.load(response)
|
||||
except (OSError, ValueError):
|
||||
return None
|
||||
|
||||
|
||||
def query_artist(session: Session, song_id: int) -> str:
|
||||
"""Find the artist name to query the APIs with.
|
||||
@dataclass(frozen=True)
|
||||
class LyricsFetchCounts:
|
||||
"""The outcome of one run of fetching the missing lyrics."""
|
||||
|
||||
:param session: The database session.
|
||||
:param song_id: The song ID.
|
||||
:return: The name of the primary-role artist with the lowest
|
||||
position.
|
||||
"""
|
||||
name: str | None = session.scalar(
|
||||
sa.select(Artist.name)
|
||||
.join(SongArtist, SongArtist.artist_id == Artist.id)
|
||||
.where(SongArtist.song_id == song_id,
|
||||
SongArtist.role == Role.PRIMARY)
|
||||
.order_by(SongArtist.position)
|
||||
.limit(1))
|
||||
assert name is not None
|
||||
return name
|
||||
fetched: int
|
||||
"""The number of songs newly fetched."""
|
||||
missed: int
|
||||
"""The number of songs every API missed."""
|
||||
|
||||
|
||||
def save_lyrics(lyrics_dir: Path, song_id: int,
|
||||
lyrics: str) -> None:
|
||||
"""Write the lyrics of a song into the cache directory.
|
||||
class LyricsFetchRunner:
|
||||
"""The orchestrator of one run of fetching missing lyrics."""
|
||||
|
||||
The cache directory is created when missing.
|
||||
__PROVENANCE_FIELDS: ClassVar[Sequence[str]] = (
|
||||
"song_id", "source", "method", "acquired_at", "note")
|
||||
"""The header columns of the lyrics provenance CSV file."""
|
||||
|
||||
The lyrics text is normalized with :func:`normalize_lyrics`
|
||||
before being written.
|
||||
@staticmethod
|
||||
def __build_normalization() -> dict[int, str | None]:
|
||||
"""Build the lyrics normalization translation table.
|
||||
|
||||
:param lyrics_dir: The lyrics cache directory.
|
||||
:param song_id: The song ID.
|
||||
:param lyrics: The lyrics text.
|
||||
:return: None.
|
||||
:raises OSError: When the file cannot be written.
|
||||
"""
|
||||
lyrics_dir.mkdir(parents=True, exist_ok=True)
|
||||
(lyrics_dir / f"{song_id}.txt").write_text(
|
||||
normalize_lyrics(lyrics), encoding="utf-8")
|
||||
:return: The codepoint-to-replacement mapping, a
|
||||
replacement of None meaning removal.
|
||||
"""
|
||||
table: dict[int, str | None] = {}
|
||||
codepoint: int
|
||||
for codepoint in range(0x80, 0xa0):
|
||||
try:
|
||||
table[codepoint] = bytes(
|
||||
[codepoint]).decode("cp1252")
|
||||
except UnicodeDecodeError:
|
||||
table[codepoint] = None
|
||||
table[0x0435] = "e"
|
||||
table[0x03cc] = "ó"
|
||||
for codepoint in (0x2005, 0x205f, 0x200a):
|
||||
table[codepoint] = " "
|
||||
for codepoint in (0x200b, 0x200c, 0x200d, 0xfeff):
|
||||
table[codepoint] = None
|
||||
return table
|
||||
|
||||
__NORMALIZATION: ClassVar[dict[int, str | None]] \
|
||||
= __build_normalization()
|
||||
"""The codepoint-to-replacement mapping applied to fetched
|
||||
lyrics: cp1252-mojibake restoration for U+0080-U+009F (with
|
||||
the five byte values undefined in cp1252 removed), homoglyph
|
||||
restoration for the Cyrillic "e" and the Greek "o" with
|
||||
tonos, ASCII-space restoration for exotic space variants, and
|
||||
removal of zero-width characters. A replacement of None
|
||||
removes the codepoint."""
|
||||
|
||||
def append_provenance(path: Path, song_id: int,
|
||||
source: str) -> None:
|
||||
"""Append a provenance row for a fetched lyrics file.
|
||||
def __init__(self, lyrics_dir: Path,
|
||||
provenance_csv: Path) -> None:
|
||||
"""Set up the fetch run.
|
||||
|
||||
The CSV file is created with the header row when missing.
|
||||
:param lyrics_dir: The lyrics cache directory.
|
||||
:param provenance_csv: The lyrics provenance CSV file.
|
||||
"""
|
||||
self.__lyrics_dir: Path = lyrics_dir
|
||||
"""The lyrics cache directory."""
|
||||
self.__provenance_csv: Path = provenance_csv
|
||||
"""The lyrics provenance CSV file."""
|
||||
self.__fetcher: LyricsFetcher = LyricsFetcher()
|
||||
"""The fetcher of the public lyrics APIs."""
|
||||
|
||||
:param path: The lyrics provenance CSV file.
|
||||
:param song_id: The song ID.
|
||||
:param source: The source name of the fetched lyrics.
|
||||
:return: None.
|
||||
:raises OSError: When the file cannot be written.
|
||||
"""
|
||||
is_new: bool = not path.exists()
|
||||
path.parent.mkdir(parents=True, exist_ok=True)
|
||||
with open(path, "a", encoding="utf-8",
|
||||
newline="") as file:
|
||||
writer: Any = csv.writer(file)
|
||||
if is_new:
|
||||
writer.writerow(PROVENANCE_FIELDS)
|
||||
writer.writerow([song_id, source, "api-fetch",
|
||||
datetime.date.today().isoformat(), ""])
|
||||
def run(self) -> LyricsFetchCounts:
|
||||
"""Fetch the missing lyrics of every song in the store.
|
||||
|
||||
Every song fetched or missed is reported on the standard
|
||||
error as an observable side effect.
|
||||
|
||||
:return: The number of songs fetched and missed.
|
||||
:raises OSError: When a cache file or the provenance CSV
|
||||
cannot be written.
|
||||
:raises sqlalchemy.exc.SQLAlchemyError: On a database
|
||||
error.
|
||||
"""
|
||||
fetched: int = 0
|
||||
missed: int = 0
|
||||
session: Session = ds.get_db()
|
||||
try:
|
||||
song: Song
|
||||
for song in session.scalars(
|
||||
sa.select(Song).order_by(Song.id)):
|
||||
if (self.__lyrics_dir
|
||||
/ f"{song.id}.txt").exists():
|
||||
continue
|
||||
if self.__fetch_one(session, song):
|
||||
fetched += 1
|
||||
else:
|
||||
missed += 1
|
||||
finally:
|
||||
session.close()
|
||||
return LyricsFetchCounts(fetched=fetched, missed=missed)
|
||||
|
||||
def __fetch_one(self, session: Session, song: Song) -> bool:
|
||||
"""Fetch and save the lyrics of one song.
|
||||
|
||||
The song is queried by its primary-role artist name; when
|
||||
every API misses and the song's full artist credit
|
||||
differs from that name, the same APIs are queried again
|
||||
with the artist credit.
|
||||
|
||||
:param session: The database session.
|
||||
:param song: The song to fetch.
|
||||
:return: True when a lyrics text was fetched and saved,
|
||||
False when every API missed on both queries.
|
||||
:raises OSError: When the cache file or the provenance
|
||||
CSV cannot be written.
|
||||
"""
|
||||
artist: str = self.__query_artist(session, song.id)
|
||||
result: tuple[str, str] | None = self.__fetcher.fetch(
|
||||
artist, song.title)
|
||||
if result is None and song.artist_credit != artist:
|
||||
result = self.__fetcher.fetch(
|
||||
song.artist_credit, song.title)
|
||||
if result is None:
|
||||
print(f"song {song.id} \"{song.title}\": miss",
|
||||
file=sys.stderr)
|
||||
return False
|
||||
lyrics: str
|
||||
source: str
|
||||
lyrics, source = result
|
||||
self.__save_lyrics(song.id, lyrics)
|
||||
self.__append_provenance(song.id, source)
|
||||
print(f"song {song.id} \"{song.title}\": {source}",
|
||||
file=sys.stderr)
|
||||
return True
|
||||
|
||||
@staticmethod
|
||||
def __query_artist(session: Session, song_id: int) -> str:
|
||||
"""Find the artist name to query the APIs with.
|
||||
|
||||
:param session: The database session.
|
||||
:param song_id: The song ID.
|
||||
:return: The name of the primary-role artist with the
|
||||
lowest position.
|
||||
"""
|
||||
name: str | None = session.scalar(
|
||||
sa.select(Artist.name)
|
||||
.join(SongArtist, SongArtist.artist_id == Artist.id)
|
||||
.where(SongArtist.song_id == song_id,
|
||||
SongArtist.role == Role.PRIMARY)
|
||||
.order_by(SongArtist.position)
|
||||
.limit(1))
|
||||
assert name is not None
|
||||
return name
|
||||
|
||||
def __save_lyrics(self, song_id: int, lyrics: str) -> None:
|
||||
"""Write the lyrics of a song into the cache directory.
|
||||
|
||||
The cache directory is created when missing.
|
||||
|
||||
The lyrics text is normalized with
|
||||
:meth:`normalize_lyrics` before being written.
|
||||
|
||||
:param song_id: The song ID.
|
||||
:param lyrics: The lyrics text.
|
||||
:return: None.
|
||||
:raises OSError: When the file cannot be written.
|
||||
"""
|
||||
self.__lyrics_dir.mkdir(parents=True, exist_ok=True)
|
||||
(self.__lyrics_dir / f"{song_id}.txt").write_text(
|
||||
self.normalize_lyrics(lyrics), encoding="utf-8")
|
||||
|
||||
def __append_provenance(self, song_id: int,
|
||||
source: str) -> None:
|
||||
"""Append a provenance row for a fetched lyrics file.
|
||||
|
||||
The CSV file is created with the header row when
|
||||
missing.
|
||||
|
||||
:param song_id: The song ID.
|
||||
:param source: The source name of the fetched lyrics.
|
||||
:return: None.
|
||||
:raises OSError: When the file cannot be written.
|
||||
"""
|
||||
is_new: bool = not self.__provenance_csv.exists()
|
||||
self.__provenance_csv.parent.mkdir(
|
||||
parents=True, exist_ok=True)
|
||||
with open(self.__provenance_csv, "a", encoding="utf-8",
|
||||
newline="") as file:
|
||||
writer: Any = csv.writer(file)
|
||||
if is_new:
|
||||
writer.writerow(self.__PROVENANCE_FIELDS)
|
||||
writer.writerow(
|
||||
[song_id, source, "api-fetch",
|
||||
datetime.date.today().isoformat(), ""])
|
||||
|
||||
@classmethod
|
||||
def normalize_lyrics(cls, text: str) -> str:
|
||||
"""Restore or remove watermark and mojibake characters.
|
||||
|
||||
:param text: The lyrics text as fetched from an API.
|
||||
:return: The text with the codepoints of the
|
||||
normalization table replaced or removed; every other
|
||||
character is unchanged.
|
||||
"""
|
||||
return text.translate(cls.__NORMALIZATION)
|
||||
|
||||
|
||||
def main(argv: list[str] | None = None) -> int:
|
||||
@@ -277,44 +372,15 @@ def main(argv: list[str] | None = None) -> int:
|
||||
"""
|
||||
started: float = time.monotonic()
|
||||
args: argparse.Namespace = parse_args(argv)
|
||||
fetcher: LyricsFetcher = LyricsFetcher()
|
||||
fetched: int = 0
|
||||
missed: int = 0
|
||||
session: Session = ds.get_db()
|
||||
try:
|
||||
song: Song
|
||||
for song in session.scalars(
|
||||
sa.select(Song).order_by(Song.id)):
|
||||
if (args.lyrics_dir / f"{song.id}.txt").exists():
|
||||
continue
|
||||
artist: str = query_artist(session, song.id)
|
||||
result: tuple[str, str] | None = fetcher.fetch(
|
||||
artist, song.title)
|
||||
if result is None and song.artist_credit != artist:
|
||||
result = fetcher.fetch(
|
||||
song.artist_credit, song.title)
|
||||
if result is None:
|
||||
missed += 1
|
||||
print(f"song {song.id} \"{song.title}\": miss",
|
||||
file=sys.stderr)
|
||||
continue
|
||||
lyrics: str
|
||||
source: str
|
||||
lyrics, source = result
|
||||
save_lyrics(args.lyrics_dir, song.id, lyrics)
|
||||
append_provenance(args.provenance_csv, song.id,
|
||||
source)
|
||||
fetched += 1
|
||||
print(f"song {song.id} \"{song.title}\": {source}",
|
||||
file=sys.stderr)
|
||||
counts: LyricsFetchCounts = LyricsFetchRunner(
|
||||
args.lyrics_dir, args.provenance_csv).run()
|
||||
except (OSError, sa.exc.SQLAlchemyError) as error:
|
||||
print(f"error: {error}", file=sys.stderr)
|
||||
return 1
|
||||
finally:
|
||||
session.close()
|
||||
attempted: int = fetched + missed
|
||||
attempted: int = counts.fetched + counts.missed
|
||||
elapsed: str = format_duration(time.monotonic() - started)
|
||||
print(f"Done. Fetched lyrics for {fetched}/{attempted}"
|
||||
f" songs. {elapsed} elapsed.",
|
||||
print(f"Done. Fetched lyrics for {counts.fetched}/"
|
||||
f"{attempted} songs. {elapsed} elapsed.",
|
||||
file=sys.stderr)
|
||||
return 0
|
||||
Reference in new issue
Block a user