Remove the missing-lyrics report from fetch-lyrics

Co-Authored-By: Claude Fable 5 <noreply@anthropic.com>
This commit is contained in:
2026-08-04 15:15:49 +08:00
co-authored by Claude Fable 5
parent 1b21471845
commit 81975b72e4
4 changed files with 13 additions and 74 deletions
+6
View File
@@ -201,3 +201,9 @@
拆成個人後,男女混合是歌曲層(song_artists+各歌手 拆成個人後,男女混合是歌曲層(song_artists+各歌手
gender)可推導的事實,不屬歌手實體。非人非團體者 gender)可推導的事實,不屬歌手實體。非人非團體者
Pinkfong)type 留空由人工判定,維持現狀。 Pinkfong)type 留空由人工判定,維持現狀。
- **fetch-lyrics 移除缺漏報表**:刪去 `missing_csv` 位置引數
與缺漏報表 CSV 輸出。理由:該報表只寫不讀(指令從未讀取
它),缺漏數字每次執行皆由工作儲存比對歌詞目錄重新算出;
worklist 角色由重跑冪等指令本身承擔;完整性已由 provenance
CSV 與 build-db 統計數字佐證;該檔案不承載任何無法重新
產生的資訊。
-1
View File
@@ -17,7 +17,6 @@ pop-fem-audit/
│ │ │ # 私人匯入腳本寫入 │ │ │ # 私人匯入腳本寫入
│ │ ├── artists-wikidata.csv # Wikidata 快照 │ │ ├── artists-wikidata.csv # Wikidata 快照
│ │ ├── lyrics_provenance.csv # 歌詞出處 │ │ ├── lyrics_provenance.csv # 歌詞出處
│ │ ├── lyrics_missing.csv # 歌詞缺漏報表
│ │ └── lyrics/ # 歌詞 .txt 快取 │ │ └── lyrics/ # 歌詞 .txt 快取
│ │ # gitignored,版權) │ │ # gitignored,版權)
│ ├── manual/ # 人工著作:只由研究者手寫 │ ├── manual/ # 人工著作:只由研究者手寫
+5 -61
View File
@@ -18,10 +18,8 @@ artist name, the same APIs are queried again with the artist
credit, to catch songs cataloged only under a joint credit such credit, to catch songs cataloged only under a joint credit such
as "Dan + Shay". as "Dan + Shay".
A song that every API misses on both queries is reported in the A song that every API misses on both queries is reported on the
missing lyrics CSV, also given as a positional command-line standard error, but does not fail the run; misses are expected.
argument, which is rewritten on every run to reflect the
current status. Misses are expected and do not fail the run.
""" """
import argparse import argparse
import csv import csv
@@ -32,7 +30,6 @@ import time
import urllib.parse import urllib.parse
import urllib.request import urllib.request
from collections.abc import Sequence from collections.abc import Sequence
from dataclasses import dataclass
from pathlib import Path from pathlib import Path
from typing import Any from typing import Any
@@ -50,9 +47,6 @@ from .models import (
PROVENANCE_FIELDS: Sequence[str] = ( PROVENANCE_FIELDS: Sequence[str] = (
"song_id", "source", "method", "acquired_at", "note") "song_id", "source", "method", "acquired_at", "note")
"""The header columns of the lyrics provenance CSV file.""" """The header columns of the lyrics provenance CSV file."""
MISSING_FIELDS: Sequence[str] = (
"song_id", "title", "artist_credit", "reason")
"""The header columns of the missing lyrics report CSV file."""
USER_AGENT: str = ("pop-fem-audit-tools" USER_AGENT: str = ("pop-fem-audit-tools"
" (https://github.com/imacat/pop-fem-audit)") " (https://github.com/imacat/pop-fem-audit)")
"""The User-Agent header sent on every HTTP request.""" """The User-Agent header sent on every HTTP request."""
@@ -62,28 +56,6 @@ SLEEP_SECONDS: float = 1.0
"""The delay between consecutive HTTP requests, in seconds.""" """The delay between consecutive HTTP requests, in seconds."""
@dataclass
class MissingLyrics:
"""One row of the missing lyrics report CSV file."""
song_id: int
"""The song ID."""
title: str
"""The song title."""
artist_credit: str
"""The artist credit of the song."""
reason: str
"""The reason the lyrics are missing."""
def to_row(self) -> list[str]:
"""Return this report entry as a CSV row.
:return: The row values, in the column order.
"""
return [str(self.song_id), self.title,
self.artist_credit, self.reason]
def parse_args(argv: list[str] | None) -> argparse.Namespace: def parse_args(argv: list[str] | None) -> argparse.Namespace:
"""Parse the command-line arguments. """Parse the command-line arguments.
@@ -100,9 +72,6 @@ def parse_args(argv: list[str] | None) -> argparse.Namespace:
parser.add_argument( parser.add_argument(
"provenance_csv", type=Path, "provenance_csv", type=Path,
help="the lyrics provenance CSV file") help="the lyrics provenance CSV file")
parser.add_argument(
"missing_csv", type=Path,
help="the missing lyrics report CSV file")
return parser.parse_args(argv) return parser.parse_args(argv)
@@ -251,27 +220,6 @@ def append_provenance(path: Path, song_id: int,
datetime.date.today().isoformat(), ""]) datetime.date.today().isoformat(), ""])
def write_missing(path: Path,
misses: Sequence[MissingLyrics]) -> None:
"""Rewrite the missing lyrics report CSV file.
The previous content is replaced, so the file reflects the
current misses only.
:param path: The missing lyrics report CSV file.
:param misses: The report entries of the songs still without
lyrics.
:return: None.
:raises OSError: When the file cannot be written.
"""
path.parent.mkdir(parents=True, exist_ok=True)
with open(path, "w", encoding="utf-8",
newline="") as file:
writer: Any = csv.writer(file)
writer.writerow(MISSING_FIELDS)
writer.writerows(x.to_row() for x in misses)
def main(argv: list[str] | None = None) -> int: def main(argv: list[str] | None = None) -> int:
"""Fetch the missing song lyrics from the public APIs. """Fetch the missing song lyrics from the public APIs.
@@ -283,7 +231,7 @@ def main(argv: list[str] | None = None) -> int:
args: argparse.Namespace = parse_args(argv) args: argparse.Namespace = parse_args(argv)
fetcher: LyricsFetcher = LyricsFetcher() fetcher: LyricsFetcher = LyricsFetcher()
fetched: int = 0 fetched: int = 0
misses: list[MissingLyrics] = [] missed: int = 0
session: Session = ds.get_db() session: Session = ds.get_db()
try: try:
song: Song song: Song
@@ -298,10 +246,7 @@ def main(argv: list[str] | None = None) -> int:
result = fetcher.fetch( result = fetcher.fetch(
song.artist_credit, song.title) song.artist_credit, song.title)
if result is None: if result is None:
misses.append(MissingLyrics( missed += 1
song_id=song.id, title=song.title,
artist_credit=song.artist_credit,
reason="not found"))
print(f"song {song.id} \"{song.title}\": miss", print(f"song {song.id} \"{song.title}\": miss",
file=sys.stderr) file=sys.stderr)
continue continue
@@ -314,12 +259,11 @@ def main(argv: list[str] | None = None) -> int:
fetched += 1 fetched += 1
print(f"song {song.id} \"{song.title}\": {source}", print(f"song {song.id} \"{song.title}\": {source}",
file=sys.stderr) file=sys.stderr)
write_missing(args.missing_csv, misses)
except (OSError, sa.exc.SQLAlchemyError) as error: except (OSError, sa.exc.SQLAlchemyError) as error:
print(f"error: {error}", file=sys.stderr) print(f"error: {error}", file=sys.stderr)
return 1 return 1
finally: finally:
session.close() session.close()
print(f"done: {fetched} fetched, {len(misses)} missed", print(f"done: {fetched} fetched, {missed} missed",
file=sys.stderr) file=sys.stderr)
return 0 return 0
+2 -12
View File
@@ -32,9 +32,6 @@ class TestFetchLyrics(unittest.TestCase):
PROVENANCE_HEADER: list[str] = [ PROVENANCE_HEADER: list[str] = [
"song_id", "source", "method", "acquired_at", "note"] "song_id", "source", "method", "acquired_at", "note"]
"""The expected header row of the provenance CSV file.""" """The expected header row of the provenance CSV file."""
MISSING_HEADER: list[str] = [
"song_id", "title", "artist_credit", "reason"]
"""The expected header row of the missing report CSV file."""
def setUp(self) -> None: def setUp(self) -> None:
"""Create a temporary capture directory with the store.""" """Create a temporary capture directory with the store."""
@@ -45,7 +42,6 @@ class TestFetchLyrics(unittest.TestCase):
self.__lyrics: Path = self.__dir / "lyrics" self.__lyrics: Path = self.__dir / "lyrics"
self.__provenance: Path = \ self.__provenance: Path = \
self.__dir / "lyrics_provenance.csv" self.__dir / "lyrics_provenance.csv"
self.__missing: Path = self.__dir / "lyrics_missing.csv"
url: str = f"sqlite:///{self.__dir}/store.sqlite3" url: str = f"sqlite:///{self.__dir}/store.sqlite3"
config.set_settings(config.Settings( config.set_settings(config.Settings(
SQLALCHEMY_DATABASE_URL=url, SQLALCHEMY_DATABASE_URL=url,
@@ -133,8 +129,7 @@ class TestFetchLyrics(unittest.TestCase):
stderr: io.StringIO = io.StringIO() stderr: io.StringIO = io.StringIO()
with redirect_stderr(stderr): with redirect_stderr(stderr):
status: int = fetch_lyrics.main( status: int = fetch_lyrics.main(
[str(self.__lyrics), str(self.__provenance), [str(self.__lyrics), str(self.__provenance)])
str(self.__missing)])
return status, stderr.getvalue() return status, stderr.getvalue()
@staticmethod @staticmethod
@@ -251,10 +246,7 @@ class TestFetchLyrics(unittest.TestCase):
self.assertEqual(status, 0) self.assertEqual(status, 0)
self.assertFalse((self.__lyrics / "1.txt").exists()) self.assertFalse((self.__lyrics / "1.txt").exists())
self.assertFalse(self.__provenance.exists()) self.assertFalse(self.__provenance.exists())
rows: list[list[str]] = self.__read_rows(self.__missing) self.assertIn("song 1 \"Hello\": miss", stderr)
self.assertEqual(len(rows), 2)
self.assertEqual(rows[0], self.MISSING_HEADER)
self.assertEqual(rows[1][:3], ["1", "Hello", "Adele"])
self.assertIn("0 fetched, 1 missed", stderr) self.assertIn("0 fetched, 1 missed", stderr)
def test_cached_song_skipped(self) -> None: def test_cached_song_skipped(self) -> None:
@@ -272,8 +264,6 @@ class TestFetchLyrics(unittest.TestCase):
(self.__lyrics / "1.txt") (self.__lyrics / "1.txt")
.read_text(encoding="utf-8"), .read_text(encoding="utf-8"),
"cached\n") "cached\n")
rows: list[list[str]] = self.__read_rows(self.__missing)
self.assertEqual(rows, [self.MISSING_HEADER])
def test_url_encoding(self) -> None: def test_url_encoding(self) -> None:
"""Test the percent-encoding of the artist and title.""" """Test the percent-encoding of the artist and title."""