Take the extra coding keywords from the command line
Co-Authored-By: Claude Opus 5 (1M context) <noreply@anthropic.com>
This commit is contained in:
+4
-2
@@ -65,8 +65,10 @@ Run、Song,一列一筆出現,依三欄排序),供收斂軌跡分析;
|
|||||||
定案詞彙表為 50 個分群組名再加上 `women-power` 一詞,共
|
定案詞彙表為 50 個分群組名再加上 `women-power` 一詞,共
|
||||||
51 個碼。`women-power` 是研究者任意決定的先驗主題(即本
|
51 個碼。`women-power` 是研究者任意決定的先驗主題(即本
|
||||||
論文的主題本身),不由資料產生,屬揭露的儀器介入。該詞
|
論文的主題本身),不由資料產生,屬揭露的儀器介入。該詞
|
||||||
以常數寫在分群子命令內、隨定案碼表一併輸出,不經人手
|
於執行時以 `--extra-keyword` 明示加入,不寫死在程式裏
|
||||||
編輯;分群明細 CSV 不含它,兩份產物因此各自誠實。
|
——研究者的介入因此每次都出現在重現命令上,而非無聲
|
||||||
|
發生;分群結果的兩份產物不含它,只有交給模型的碼表含
|
||||||
|
它。
|
||||||
|
|
||||||
注入而非另設篩選軌的理由:讓研究者的主題詞與模型自己
|
注入而非另設篩選軌的理由:讓研究者的主題詞與模型自己
|
||||||
收斂出的類別(分群已自行長出 `female-empowerment` 等組)
|
收斂出的類別(分群已自行長出 `female-empowerment` 等組)
|
||||||
|
|||||||
@@ -22,9 +22,12 @@ result alone, as :data:`RESULT_GROUPS_CSV`. The group name
|
|||||||
keywords alone are written as a text file, one per line, as
|
keywords alone are written as a text file, one per line, as
|
||||||
:data:`RESULT_KEYWORDS_TXT`. The coding keyword set for
|
:data:`RESULT_KEYWORDS_TXT`. The coding keyword set for
|
||||||
``export-llm-input --extras`` is written as a JSON file holding the
|
``export-llm-input --extras`` is written as a JSON file holding the
|
||||||
group name keywords plus the researcher's a-priori topic term (see
|
group name keywords plus every extra a-priori keyword the caller
|
||||||
:data:`EXTRA_KEYWORD`), as :data:`KEYWORDS_TO_MERGE_JSON`. The
|
gives with the repeatable ``--extra-keyword`` command-line option,
|
||||||
step is fully deterministic; no LLM call is made.
|
as :data:`KEYWORDS_TO_MERGE_JSON`; with no ``--extra-keyword``, it
|
||||||
|
holds the group names alone. No default extra keyword is ever
|
||||||
|
injected; the caller supplies each one consciously. The step is
|
||||||
|
fully deterministic; no LLM call is made.
|
||||||
"""
|
"""
|
||||||
import argparse
|
import argparse
|
||||||
import csv
|
import csv
|
||||||
@@ -44,9 +47,6 @@ CLUSTER_EXTRA_MESSAGE: str = (
|
|||||||
" pip install -e \"tools/[cluster]\"")
|
" pip install -e \"tools/[cluster]\"")
|
||||||
"""The error message shown when the heavy clustering dependencies
|
"""The error message shown when the heavy clustering dependencies
|
||||||
are not installed."""
|
are not installed."""
|
||||||
EXTRA_KEYWORD: str = "women-power"
|
|
||||||
"""The researcher's a-priori topic term, included in the coding
|
|
||||||
keyword set although it is not a clustering result."""
|
|
||||||
SOURCE_KEYWORDS_TXT: str = "source-keywords.txt"
|
SOURCE_KEYWORDS_TXT: str = "source-keywords.txt"
|
||||||
"""The pooled keyword text file's fixed name under the output
|
"""The pooled keyword text file's fixed name under the output
|
||||||
directory."""
|
directory."""
|
||||||
@@ -103,6 +103,14 @@ def parse_args(argv: list[str] | None) -> argparse.Namespace:
|
|||||||
parser.add_argument(
|
parser.add_argument(
|
||||||
"--clusters", type=int, default=DEFAULT_CLUSTERS,
|
"--clusters", type=int, default=DEFAULT_CLUSTERS,
|
||||||
help=f"the number of clusters (default {DEFAULT_CLUSTERS})")
|
help=f"the number of clusters (default {DEFAULT_CLUSTERS})")
|
||||||
|
parser.add_argument(
|
||||||
|
"--extra-keyword", dest="extra_keywords", action="append",
|
||||||
|
default=None,
|
||||||
|
help="an extra a-priori keyword to add to"
|
||||||
|
f" {KEYWORDS_TO_MERGE_JSON} alongside the clustered"
|
||||||
|
" group names; repeatable; may be given any number"
|
||||||
|
" of times; not added to any other output file"
|
||||||
|
" (default: none)")
|
||||||
return parser.parse_args(argv)
|
return parser.parse_args(argv)
|
||||||
|
|
||||||
|
|
||||||
@@ -363,7 +371,7 @@ def write_groups(path: Path, groups: dict[str, list[str]]) -> None:
|
|||||||
row per member keyword. Rows are sorted by group name
|
row per member keyword. Rows are sorted by group name
|
||||||
lexicographically, then by keyword lexicographically. The
|
lexicographically, then by keyword lexicographically. The
|
||||||
file records the clustering result alone; it holds no row for
|
file records the clustering result alone; it holds no row for
|
||||||
:data:`EXTRA_KEYWORD`.
|
any ``--extra-keyword`` given on the command line.
|
||||||
|
|
||||||
:param path: The path of the group membership CSV file to
|
:param path: The path of the group membership CSV file to
|
||||||
write.
|
write.
|
||||||
@@ -389,7 +397,8 @@ def write_keyword_names(path: Path,
|
|||||||
Writes a text file holding the lexicographically sorted
|
Writes a text file holding the lexicographically sorted
|
||||||
group names, one per line, UTF-8, LF line endings, with a
|
group names, one per line, UTF-8, LF line endings, with a
|
||||||
trailing newline. The file records the clustering result
|
trailing newline. The file records the clustering result
|
||||||
alone; it holds no :data:`EXTRA_KEYWORD` line.
|
alone; it holds no line for any ``--extra-keyword`` given on
|
||||||
|
the command line.
|
||||||
|
|
||||||
:param path: The path of the keyword text file to write.
|
:param path: The path of the keyword text file to write.
|
||||||
:param groups: The keyword members of every group, keyed by
|
:param groups: The keyword members of every group, keyed by
|
||||||
@@ -402,24 +411,56 @@ def write_keyword_names(path: Path,
|
|||||||
"".join(f"{x}\n" for x in names), encoding="utf-8")
|
"".join(f"{x}\n" for x in names), encoding="utf-8")
|
||||||
|
|
||||||
|
|
||||||
def write_keywords_to_merge(path: Path,
|
def validate_extra_keywords(
|
||||||
|
extra_keywords: list[str],
|
||||||
groups: dict[str, list[str]]) -> None:
|
groups: dict[str, list[str]]) -> None:
|
||||||
|
"""Validate the extra keywords given on the command line.
|
||||||
|
|
||||||
|
:param extra_keywords: The extra keywords given via
|
||||||
|
``--extra-keyword``, in the given order.
|
||||||
|
:param groups: The keyword members of every group, keyed by
|
||||||
|
the group's medoid name.
|
||||||
|
:return: None.
|
||||||
|
:raises ValueError: When an extra keyword duplicates a
|
||||||
|
clustered group name or another extra keyword.
|
||||||
|
"""
|
||||||
|
seen: set[str] = set()
|
||||||
|
keyword: str
|
||||||
|
for keyword in extra_keywords:
|
||||||
|
if keyword in groups:
|
||||||
|
raise ValueError(
|
||||||
|
f"extra keyword \"{keyword}\" duplicates a"
|
||||||
|
" clustered group name")
|
||||||
|
if keyword in seen:
|
||||||
|
raise ValueError(
|
||||||
|
f"extra keyword \"{keyword}\" given more than"
|
||||||
|
" once")
|
||||||
|
seen.add(keyword)
|
||||||
|
|
||||||
|
|
||||||
|
def write_keywords_to_merge(path: Path,
|
||||||
|
groups: dict[str, list[str]],
|
||||||
|
extra_keywords: list[str]) -> None:
|
||||||
"""Write the coding keyword set JSON file.
|
"""Write the coding keyword set JSON file.
|
||||||
|
|
||||||
Writes a JSON file holding a single object with one
|
Writes a JSON file holding a single object with one
|
||||||
``keywords`` key, whose value is the lexicographically
|
``keywords`` key, whose value is the lexicographically
|
||||||
sorted list of the group names plus :data:`EXTRA_KEYWORD`,
|
sorted list of the group names plus every given extra
|
||||||
UTF-8, with a trailing newline. This is the file
|
keyword, UTF-8, with a trailing newline. With no extra
|
||||||
``export-llm-input --extras`` consumes.
|
keyword, the list holds the group names alone. This is the
|
||||||
|
file ``export-llm-input --extras`` consumes.
|
||||||
|
|
||||||
:param path: The path of the keyword JSON file to write.
|
:param path: The path of the keyword JSON file to write.
|
||||||
:param groups: The keyword members of every group, keyed by
|
:param groups: The keyword members of every group, keyed by
|
||||||
the group's medoid name.
|
the group's medoid name.
|
||||||
|
:param extra_keywords: The extra a-priori keywords given via
|
||||||
|
``--extra-keyword``, to include alongside the group
|
||||||
|
names.
|
||||||
:return: None.
|
:return: None.
|
||||||
:raises OSError: When the file cannot be written.
|
:raises OSError: When the file cannot be written.
|
||||||
"""
|
"""
|
||||||
keywords: list[str] = sorted(
|
keywords: list[str] = sorted(
|
||||||
[*groups.keys(), EXTRA_KEYWORD])
|
[*groups.keys(), *extra_keywords])
|
||||||
data: dict[str, list[str]] = {"keywords": keywords}
|
data: dict[str, list[str]] = {"keywords": keywords}
|
||||||
path.write_text(
|
path.write_text(
|
||||||
json.dumps(data, ensure_ascii=False, indent=1) + "\n",
|
json.dumps(data, ensure_ascii=False, indent=1) + "\n",
|
||||||
@@ -436,8 +477,10 @@ def main(argv: list[str] | None = None) -> int:
|
|||||||
clustering result alone; the group name keyword text file,
|
clustering result alone; the group name keyword text file,
|
||||||
holding the same group names as a readable list; and the
|
holding the same group names as a readable list; and the
|
||||||
coding keyword set JSON file, holding the group names plus
|
coding keyword set JSON file, holding the group names plus
|
||||||
:data:`EXTRA_KEYWORD`. When the input is rejected, none of
|
every extra keyword given via ``--extra-keyword``. When the
|
||||||
the five files is written.
|
input is rejected, or an extra keyword duplicates a group
|
||||||
|
name or another extra keyword, none of the five files is
|
||||||
|
written.
|
||||||
|
|
||||||
:param argv: The command-line arguments, or None for
|
:param argv: The command-line arguments, or None for
|
||||||
``sys.argv``.
|
``sys.argv``.
|
||||||
@@ -465,6 +508,12 @@ def main(argv: list[str] | None = None) -> int:
|
|||||||
except (RuntimeError, ValueError) as error:
|
except (RuntimeError, ValueError) as error:
|
||||||
print(f"error: {error}", file=sys.stderr)
|
print(f"error: {error}", file=sys.stderr)
|
||||||
return 1
|
return 1
|
||||||
|
extra_keywords: list[str] = args.extra_keywords or []
|
||||||
|
try:
|
||||||
|
validate_extra_keywords(extra_keywords, groups)
|
||||||
|
except ValueError as error:
|
||||||
|
print(f"error: {error}", file=sys.stderr)
|
||||||
|
return 1
|
||||||
args.output_dir.mkdir(parents=True, exist_ok=True)
|
args.output_dir.mkdir(parents=True, exist_ok=True)
|
||||||
write_pool(
|
write_pool(
|
||||||
args.output_dir / SOURCE_KEYWORDS_TXT, keywords)
|
args.output_dir / SOURCE_KEYWORDS_TXT, keywords)
|
||||||
@@ -474,7 +523,8 @@ def main(argv: list[str] | None = None) -> int:
|
|||||||
write_keyword_names(
|
write_keyword_names(
|
||||||
args.output_dir / RESULT_KEYWORDS_TXT, groups)
|
args.output_dir / RESULT_KEYWORDS_TXT, groups)
|
||||||
write_keywords_to_merge(
|
write_keywords_to_merge(
|
||||||
args.output_dir / KEYWORDS_TO_MERGE_JSON, groups)
|
args.output_dir / KEYWORDS_TO_MERGE_JSON, groups,
|
||||||
|
extra_keywords)
|
||||||
elapsed: str = format_duration(time.monotonic() - started)
|
elapsed: str = format_duration(time.monotonic() - started)
|
||||||
print(
|
print(
|
||||||
f"done: {len(keywords)} keywords pooled from"
|
f"done: {len(keywords)} keywords pooled from"
|
||||||
|
|||||||
@@ -184,7 +184,7 @@ class TestClusterKeywords(unittest.TestCase):
|
|||||||
def __read_keywords_to_merge(self) -> list[str]:
|
def __read_keywords_to_merge(self) -> list[str]:
|
||||||
"""Read the coding keyword set JSON file.
|
"""Read the coding keyword set JSON file.
|
||||||
|
|
||||||
:return: The group names plus :data:`EXTRA_KEYWORD`
|
:return: The group names plus every given extra keyword,
|
||||||
under the "keywords" key.
|
under the "keywords" key.
|
||||||
"""
|
"""
|
||||||
data: dict[str, list[str]] = json.loads(
|
data: dict[str, list[str]] = json.loads(
|
||||||
@@ -391,8 +391,7 @@ class TestClusterKeywords(unittest.TestCase):
|
|||||||
|
|
||||||
def test_keywords_txt_sorted_medoids(self) -> None:
|
def test_keywords_txt_sorted_medoids(self) -> None:
|
||||||
"""Test that the result keyword text file holds the
|
"""Test that the result keyword text file holds the
|
||||||
sorted medoid group names without the extra a-priori
|
sorted medoid group names alone."""
|
||||||
keyword."""
|
|
||||||
self.__write_output(self.__run1, [
|
self.__write_output(self.__run1, [
|
||||||
{"id": "song-1", "text": json.dumps(
|
{"id": "song-1", "text": json.dumps(
|
||||||
{"a-left": 1, "a-center": 1, "a-right": 1})},
|
{"a-left": 1, "a-center": 1, "a-right": 1})},
|
||||||
@@ -407,12 +406,12 @@ class TestClusterKeywords(unittest.TestCase):
|
|||||||
names: list[str] = self.__read_result_keywords()
|
names: list[str] = self.__read_result_keywords()
|
||||||
self.assertEqual(names, ["a-center", "b-middle"])
|
self.assertEqual(names, ["a-center", "b-middle"])
|
||||||
self.assertEqual(names, sorted(names))
|
self.assertEqual(names, sorted(names))
|
||||||
self.assertNotIn(cluster_keywords.EXTRA_KEYWORD, names)
|
|
||||||
|
|
||||||
def test_keywords_to_merge_json_sorted_medoids(self) -> None:
|
def test_no_extra_keyword_merge_json_equals_group_names(
|
||||||
"""Test that the coding keyword set JSON file holds the
|
self) -> None:
|
||||||
sorted medoid group names plus the extra a-priori
|
"""Test that with no ``--extra-keyword`` given, the
|
||||||
keyword."""
|
coding keyword set JSON file holds exactly the sorted
|
||||||
|
medoid group names."""
|
||||||
self.__write_output(self.__run1, [
|
self.__write_output(self.__run1, [
|
||||||
{"id": "song-1", "text": json.dumps(
|
{"id": "song-1", "text": json.dumps(
|
||||||
{"a-left": 1, "a-center": 1, "a-right": 1})},
|
{"a-left": 1, "a-center": 1, "a-right": 1})},
|
||||||
@@ -425,15 +424,14 @@ class TestClusterKeywords(unittest.TestCase):
|
|||||||
status, _ = self.__run_cluster()
|
status, _ = self.__run_cluster()
|
||||||
self.assertEqual(status, 0)
|
self.assertEqual(status, 0)
|
||||||
keywords: list[str] = self.__read_keywords_to_merge()
|
keywords: list[str] = self.__read_keywords_to_merge()
|
||||||
self.assertEqual(
|
self.assertEqual(keywords, ["a-center", "b-middle"])
|
||||||
keywords,
|
|
||||||
["a-center", "b-middle", cluster_keywords.EXTRA_KEYWORD])
|
|
||||||
self.assertEqual(keywords, sorted(keywords))
|
self.assertEqual(keywords, sorted(keywords))
|
||||||
self.assertEqual(len(keywords), 2 + 1)
|
|
||||||
|
|
||||||
def test_extra_keyword_absent_from_groups_csv(self) -> None:
|
def test_extra_keyword_included_in_merge_json(self) -> None:
|
||||||
"""Test that the extra a-priori keyword appears in no row
|
"""Test that a single ``--extra-keyword`` is added to the
|
||||||
of the group membership CSV file."""
|
coding keyword set JSON file, sorted with the group
|
||||||
|
names, while the group name text file and the group
|
||||||
|
membership CSV file hold no trace of it."""
|
||||||
self.__write_output(self.__run1, [
|
self.__write_output(self.__run1, [
|
||||||
{"id": "song-1", "text": json.dumps(
|
{"id": "song-1", "text": json.dumps(
|
||||||
{"a-left": 1, "a-center": 1, "a-right": 1})},
|
{"a-left": 1, "a-center": 1, "a-right": 1})},
|
||||||
@@ -443,8 +441,88 @@ class TestClusterKeywords(unittest.TestCase):
|
|||||||
{"b-north": 1, "b-middle": 1, "b-south": 1})},
|
{"b-north": 1, "b-middle": 1, "b-south": 1})},
|
||||||
])
|
])
|
||||||
status: int
|
status: int
|
||||||
status, _ = self.__run_cluster()
|
status, _ = self.__run_cluster(
|
||||||
|
extra_args=["--extra-keyword", "zzz-extra"])
|
||||||
self.assertEqual(status, 0)
|
self.assertEqual(status, 0)
|
||||||
|
self.assertEqual(
|
||||||
|
self.__read_keywords_to_merge(),
|
||||||
|
["a-center", "b-middle", "zzz-extra"])
|
||||||
|
self.assertEqual(
|
||||||
|
self.__read_result_keywords(),
|
||||||
|
["a-center", "b-middle"])
|
||||||
rows: list[list[str]] = self.__read_groups()
|
rows: list[list[str]] = self.__read_groups()
|
||||||
|
row: list[str]
|
||||||
for row in rows:
|
for row in rows:
|
||||||
self.assertNotIn(cluster_keywords.EXTRA_KEYWORD, row)
|
self.assertNotIn("zzz-extra", row)
|
||||||
|
|
||||||
|
def test_multiple_extra_keywords_sorted_union(self) -> None:
|
||||||
|
"""Test that several ``--extra-keyword`` options are all
|
||||||
|
added, and the whole coding keyword set is
|
||||||
|
lexicographically sorted."""
|
||||||
|
self.__write_output(self.__run1, [
|
||||||
|
{"id": "song-1", "text": json.dumps(
|
||||||
|
{"a-left": 1, "a-center": 1, "a-right": 1})},
|
||||||
|
])
|
||||||
|
self.__write_output(self.__run2, [
|
||||||
|
{"id": "song-2", "text": json.dumps(
|
||||||
|
{"b-north": 1, "b-middle": 1, "b-south": 1})},
|
||||||
|
])
|
||||||
|
status: int
|
||||||
|
status, _ = self.__run_cluster(extra_args=[
|
||||||
|
"--extra-keyword", "zzz-extra",
|
||||||
|
"--extra-keyword", "aaa-extra"])
|
||||||
|
self.assertEqual(status, 0)
|
||||||
|
keywords: list[str] = self.__read_keywords_to_merge()
|
||||||
|
self.assertEqual(
|
||||||
|
keywords,
|
||||||
|
["a-center", "aaa-extra", "b-middle", "zzz-extra"])
|
||||||
|
self.assertEqual(keywords, sorted(keywords))
|
||||||
|
|
||||||
|
def test_duplicate_extra_keyword_rejected(self) -> None:
|
||||||
|
"""Test that repeating the same ``--extra-keyword`` value
|
||||||
|
fails the run without writing any output file."""
|
||||||
|
self.__write_output(self.__run1, [
|
||||||
|
{"id": "song-1", "text": json.dumps(
|
||||||
|
{"a-left": 1, "a-center": 1, "a-right": 1})},
|
||||||
|
])
|
||||||
|
self.__write_output(self.__run2, [
|
||||||
|
{"id": "song-2", "text": json.dumps(
|
||||||
|
{"b-north": 1, "b-middle": 1, "b-south": 1})},
|
||||||
|
])
|
||||||
|
status: int
|
||||||
|
stderr: str
|
||||||
|
status, stderr = self.__run_cluster(extra_args=[
|
||||||
|
"--extra-keyword", "zzz-extra",
|
||||||
|
"--extra-keyword", "zzz-extra"])
|
||||||
|
self.assertEqual(status, 1)
|
||||||
|
self.assertIn("zzz-extra", stderr)
|
||||||
|
self.assertFalse(self.__source_keywords_txt.exists())
|
||||||
|
self.assertFalse(self.__source_provenance_csv.exists())
|
||||||
|
self.assertFalse(self.__result_groups_csv.exists())
|
||||||
|
self.assertFalse(self.__result_keywords_txt.exists())
|
||||||
|
self.assertFalse(self.__keywords_to_merge_json.exists())
|
||||||
|
|
||||||
|
def test_extra_keyword_duplicating_group_name_rejected(
|
||||||
|
self) -> None:
|
||||||
|
"""Test that an ``--extra-keyword`` matching a clustered
|
||||||
|
group name fails the run without writing any output
|
||||||
|
file."""
|
||||||
|
self.__write_output(self.__run1, [
|
||||||
|
{"id": "song-1", "text": json.dumps(
|
||||||
|
{"a-left": 1, "a-center": 1, "a-right": 1})},
|
||||||
|
])
|
||||||
|
self.__write_output(self.__run2, [
|
||||||
|
{"id": "song-2", "text": json.dumps(
|
||||||
|
{"b-north": 1, "b-middle": 1, "b-south": 1})},
|
||||||
|
])
|
||||||
|
status: int
|
||||||
|
stderr: str
|
||||||
|
status, stderr = self.__run_cluster(
|
||||||
|
extra_args=["--extra-keyword", "a-center"])
|
||||||
|
self.assertEqual(status, 1)
|
||||||
|
self.assertIn("a-center", stderr)
|
||||||
|
self.assertFalse(self.__source_keywords_txt.exists())
|
||||||
|
self.assertFalse(self.__source_provenance_csv.exists())
|
||||||
|
self.assertFalse(self.__result_groups_csv.exists())
|
||||||
|
self.assertFalse(self.__result_keywords_txt.exists())
|
||||||
|
self.assertFalse(self.__keywords_to_merge_json.exists())
|
||||||
|
|||||||
Reference in New Issue
Block a user