Compare commits
122
Commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
353ff02b1f | ||
|
|
e9fd56e2a8 | ||
|
|
f60f6b218a | ||
|
|
ada85291fe | ||
|
|
9b4ee3a658 | ||
|
|
2b2a93fccf | ||
|
|
a431520f97 | ||
|
|
cf00240b72 | ||
|
|
a3af04cd93 | ||
|
|
341f0ccd58 | ||
|
|
3b2a15721c | ||
|
|
e131aa4f72 | ||
|
|
ab7d51d5d7 | ||
|
|
72fadd5b40 | ||
|
|
88d64f0553 | ||
|
|
5961a739bd | ||
|
|
9c0a9c80c8 | ||
|
|
d9e3d5148f | ||
|
|
9756c07cad | ||
|
|
e9c29e5867 | ||
|
|
171d2f81fe | ||
|
|
df363e14f0 | ||
|
|
d591931101 | ||
|
|
03ee51912d | ||
|
|
87ad31568a | ||
|
|
ed54a1248f | ||
|
|
963504e9c3 | ||
|
|
64dd9b0cc3 | ||
|
|
41d2822063 | ||
|
|
70578b5d77 | ||
|
|
ff2a3a217f | ||
|
|
2c15c0bb8b | ||
|
|
5d10383dcb | ||
|
|
99f5c1c260 | ||
|
|
d370ac51dc | ||
|
|
c7b630c632 | ||
|
|
5194d4246b | ||
|
|
36b3456414 | ||
|
|
511abd0a55 | ||
|
|
047f322bf3 | ||
|
|
2d8a34ea9e | ||
|
|
1c78f4a918 | ||
|
|
71f9c713f0 | ||
|
|
95b843f76e | ||
|
|
76d5df3a87 | ||
|
|
a21e407e2f | ||
|
|
7a5176b8b3 | ||
|
|
045086f9ca | ||
|
|
87352a2530 | ||
|
|
bfb5208133 | ||
|
|
5018f12cd1 | ||
|
|
10fccca108 | ||
|
|
4dd82867fa | ||
|
|
8c06406b3a | ||
|
|
3a2419f43a | ||
|
|
8299906b51 | ||
|
|
adbd9c1a24 | ||
|
|
94971eb07d | ||
|
|
85db4dac12 | ||
|
|
c2ffa9d247 | ||
|
|
c997cc8d74 | ||
|
|
c1e099901d | ||
|
|
787ca10f29 | ||
|
|
1db46039fe | ||
|
|
d4c313734a | ||
|
|
fbb80180af | ||
|
|
def2affa95 | ||
|
|
c5ad09524b | ||
|
|
5b376e001d | ||
|
|
59f62cd4d4 | ||
|
|
864735352a | ||
|
|
1c4d5c5ed7 | ||
|
|
066ae15800 | ||
|
|
bf2b4970fd | ||
|
|
896ee09b21 | ||
|
|
a14e2a878a | ||
|
|
b0a79af1a8 | ||
|
|
0b995e5a51 | ||
|
|
4eb278164b | ||
|
|
d9014cf1f3 | ||
|
|
4a0a4ceead | ||
|
|
2e5ecaaad0 | ||
|
|
e858f7a38f | ||
|
|
5050b16812 | ||
|
|
93b8afc547 | ||
|
|
cba73a6ace | ||
|
|
bc98fdf171 | ||
|
|
9ef84cfe54 | ||
|
|
09a8381519 | ||
|
|
cf80a6f51e | ||
|
|
0b77310d86 | ||
|
|
5092e04d34 | ||
|
|
376916fb8c | ||
|
|
ce8a217e94 | ||
|
|
14cc80e662 | ||
|
|
21c5ae5481 | ||
|
|
383670ab2b | ||
|
|
c59a023095 | ||
|
|
10dc4cb693 | ||
|
|
449675e7a2 | ||
|
|
a7806c0c9d | ||
|
|
105664e93b | ||
|
|
1220c21807 | ||
|
|
86914de850 | ||
|
|
8463cfc906 | ||
|
|
ef4112c61e | ||
|
|
1eaebeb5fb | ||
|
|
6fc4b7db17 | ||
|
|
c94b6f2589 | ||
|
|
b3ac12c88d | ||
|
|
5b2c192827 | ||
|
|
991ab5d93a | ||
|
|
78cf5e6ce0 | ||
|
|
aa78a54d12 | ||
|
|
3d472d5c15 | ||
|
|
2f048dee36 | ||
|
|
a3062e848c | ||
|
|
1b2d46cb42 | ||
|
|
f2ccca60b5 | ||
|
|
db7b5f6679 | ||
|
|
98b665537f | ||
|
|
9a089f6c6b |
@@ -0,0 +1,396 @@
|
|||||||
|
Attribution 4.0 International
|
||||||
|
|
||||||
|
=======================================================================
|
||||||
|
|
||||||
|
Creative Commons Corporation ("Creative Commons") is not a law firm and
|
||||||
|
does not provide legal services or legal advice. Distribution of
|
||||||
|
Creative Commons public licenses does not create a lawyer-client or
|
||||||
|
other relationship. Creative Commons makes its licenses and related
|
||||||
|
information available on an "as-is" basis. Creative Commons gives no
|
||||||
|
warranties regarding its licenses, any material licensed under their
|
||||||
|
terms and conditions, or any related information. Creative Commons
|
||||||
|
disclaims all liability for damages resulting from their use to the
|
||||||
|
fullest extent possible.
|
||||||
|
|
||||||
|
Using Creative Commons Public Licenses
|
||||||
|
|
||||||
|
Creative Commons public licenses provide a standard set of terms and
|
||||||
|
conditions that creators and other rights holders may use to share
|
||||||
|
original works of authorship and other material subject to copyright
|
||||||
|
and certain other rights specified in the public license below. The
|
||||||
|
following considerations are for informational purposes only, are not
|
||||||
|
exhaustive, and do not form part of our licenses.
|
||||||
|
|
||||||
|
Considerations for licensors: Our public licenses are
|
||||||
|
intended for use by those authorized to give the public
|
||||||
|
permission to use material in ways otherwise restricted by
|
||||||
|
copyright and certain other rights. Our licenses are
|
||||||
|
irrevocable. Licensors should read and understand the terms
|
||||||
|
and conditions of the license they choose before applying it.
|
||||||
|
Licensors should also secure all rights necessary before
|
||||||
|
applying our licenses so that the public can reuse the
|
||||||
|
material as expected. Licensors should clearly mark any
|
||||||
|
material not subject to the license. This includes other CC-
|
||||||
|
licensed material, or material used under an exception or
|
||||||
|
limitation to copyright. More considerations for licensors:
|
||||||
|
wiki.creativecommons.org/Considerations_for_licensors
|
||||||
|
|
||||||
|
Considerations for the public: By using one of our public
|
||||||
|
licenses, a licensor grants the public permission to use the
|
||||||
|
licensed material under specified terms and conditions. If
|
||||||
|
the licensor's permission is not necessary for any reason--for
|
||||||
|
example, because of any applicable exception or limitation to
|
||||||
|
copyright--then that use is not regulated by the license. Our
|
||||||
|
licenses grant only permissions under copyright and certain
|
||||||
|
other rights that a licensor has authority to grant. Use of
|
||||||
|
the licensed material may still be restricted for other
|
||||||
|
reasons, including because others have copyright or other
|
||||||
|
rights in the material. A licensor may make special requests,
|
||||||
|
such as asking that all changes be marked or described.
|
||||||
|
Although not required by our licenses, you are encouraged to
|
||||||
|
respect those requests where reasonable. More considerations
|
||||||
|
for the public:
|
||||||
|
wiki.creativecommons.org/Considerations_for_licensees
|
||||||
|
|
||||||
|
=======================================================================
|
||||||
|
|
||||||
|
Creative Commons Attribution 4.0 International Public License
|
||||||
|
|
||||||
|
By exercising the Licensed Rights (defined below), You accept and agree
|
||||||
|
to be bound by the terms and conditions of this Creative Commons
|
||||||
|
Attribution 4.0 International Public License ("Public License"). To the
|
||||||
|
extent this Public License may be interpreted as a contract, You are
|
||||||
|
granted the Licensed Rights in consideration of Your acceptance of
|
||||||
|
these terms and conditions, and the Licensor grants You such rights in
|
||||||
|
consideration of benefits the Licensor receives from making the
|
||||||
|
Licensed Material available under these terms and conditions.
|
||||||
|
|
||||||
|
|
||||||
|
Section 1 -- Definitions.
|
||||||
|
|
||||||
|
a. Adapted Material means material subject to Copyright and Similar
|
||||||
|
Rights that is derived from or based upon the Licensed Material
|
||||||
|
and in which the Licensed Material is translated, altered,
|
||||||
|
arranged, transformed, or otherwise modified in a manner requiring
|
||||||
|
permission under the Copyright and Similar Rights held by the
|
||||||
|
Licensor. For purposes of this Public License, where the Licensed
|
||||||
|
Material is a musical work, performance, or sound recording,
|
||||||
|
Adapted Material is always produced where the Licensed Material is
|
||||||
|
synched in timed relation with a moving image.
|
||||||
|
|
||||||
|
b. Adapter's License means the license You apply to Your Copyright
|
||||||
|
and Similar Rights in Your contributions to Adapted Material in
|
||||||
|
accordance with the terms and conditions of this Public License.
|
||||||
|
|
||||||
|
c. Copyright and Similar Rights means copyright and/or similar rights
|
||||||
|
closely related to copyright including, without limitation,
|
||||||
|
performance, broadcast, sound recording, and Sui Generis Database
|
||||||
|
Rights, without regard to how the rights are labeled or
|
||||||
|
categorized. For purposes of this Public License, the rights
|
||||||
|
specified in Section 2(b)(1)-(2) are not Copyright and Similar
|
||||||
|
Rights.
|
||||||
|
|
||||||
|
d. Effective Technological Measures means those measures that, in the
|
||||||
|
absence of proper authority, may not be circumvented under laws
|
||||||
|
fulfilling obligations under Article 11 of the WIPO Copyright
|
||||||
|
Treaty adopted on December 20, 1996, and/or similar international
|
||||||
|
agreements.
|
||||||
|
|
||||||
|
e. Exceptions and Limitations means fair use, fair dealing, and/or
|
||||||
|
any other exception or limitation to Copyright and Similar Rights
|
||||||
|
that applies to Your use of the Licensed Material.
|
||||||
|
|
||||||
|
f. Licensed Material means the artistic or literary work, database,
|
||||||
|
or other material to which the Licensor applied this Public
|
||||||
|
License.
|
||||||
|
|
||||||
|
g. Licensed Rights means the rights granted to You subject to the
|
||||||
|
terms and conditions of this Public License, which are limited to
|
||||||
|
all Copyright and Similar Rights that apply to Your use of the
|
||||||
|
Licensed Material and that the Licensor has authority to license.
|
||||||
|
|
||||||
|
h. Licensor means the individual(s) or entity(ies) granting rights
|
||||||
|
under this Public License.
|
||||||
|
|
||||||
|
i. Share means to provide material to the public by any means or
|
||||||
|
process that requires permission under the Licensed Rights, such
|
||||||
|
as reproduction, public display, public performance, distribution,
|
||||||
|
dissemination, communication, or importation, and to make material
|
||||||
|
available to the public including in ways that members of the
|
||||||
|
public may access the material from a place and at a time
|
||||||
|
individually chosen by them.
|
||||||
|
|
||||||
|
j. Sui Generis Database Rights means rights other than copyright
|
||||||
|
resulting from Directive 96/9/EC of the European Parliament and of
|
||||||
|
the Council of 11 March 1996 on the legal protection of databases,
|
||||||
|
as amended and/or succeeded, as well as other essentially
|
||||||
|
equivalent rights anywhere in the world.
|
||||||
|
|
||||||
|
k. You means the individual or entity exercising the Licensed Rights
|
||||||
|
under this Public License. Your has a corresponding meaning.
|
||||||
|
|
||||||
|
|
||||||
|
Section 2 -- Scope.
|
||||||
|
|
||||||
|
a. License grant.
|
||||||
|
|
||||||
|
1. Subject to the terms and conditions of this Public License,
|
||||||
|
the Licensor hereby grants You a worldwide, royalty-free,
|
||||||
|
non-sublicensable, non-exclusive, irrevocable license to
|
||||||
|
exercise the Licensed Rights in the Licensed Material to:
|
||||||
|
|
||||||
|
a. reproduce and Share the Licensed Material, in whole or
|
||||||
|
in part; and
|
||||||
|
|
||||||
|
b. produce, reproduce, and Share Adapted Material.
|
||||||
|
|
||||||
|
2. Exceptions and Limitations. For the avoidance of doubt, where
|
||||||
|
Exceptions and Limitations apply to Your use, this Public
|
||||||
|
License does not apply, and You do not need to comply with
|
||||||
|
its terms and conditions.
|
||||||
|
|
||||||
|
3. Term. The term of this Public License is specified in Section
|
||||||
|
6(a).
|
||||||
|
|
||||||
|
4. Media and formats; technical modifications allowed. The
|
||||||
|
Licensor authorizes You to exercise the Licensed Rights in
|
||||||
|
all media and formats whether now known or hereafter created,
|
||||||
|
and to make technical modifications necessary to do so. The
|
||||||
|
Licensor waives and/or agrees not to assert any right or
|
||||||
|
authority to forbid You from making technical modifications
|
||||||
|
necessary to exercise the Licensed Rights, including
|
||||||
|
technical modifications necessary to circumvent Effective
|
||||||
|
Technological Measures. For purposes of this Public License,
|
||||||
|
simply making modifications authorized by this Section 2(a)
|
||||||
|
(4) never produces Adapted Material.
|
||||||
|
|
||||||
|
5. Downstream recipients.
|
||||||
|
|
||||||
|
a. Offer from the Licensor -- Licensed Material. Every
|
||||||
|
recipient of the Licensed Material automatically
|
||||||
|
receives an offer from the Licensor to exercise the
|
||||||
|
Licensed Rights under the terms and conditions of this
|
||||||
|
Public License.
|
||||||
|
|
||||||
|
b. No downstream restrictions. You may not offer or impose
|
||||||
|
any additional or different terms or conditions on, or
|
||||||
|
apply any Effective Technological Measures to, the
|
||||||
|
Licensed Material if doing so restricts exercise of the
|
||||||
|
Licensed Rights by any recipient of the Licensed
|
||||||
|
Material.
|
||||||
|
|
||||||
|
6. No endorsement. Nothing in this Public License constitutes or
|
||||||
|
may be construed as permission to assert or imply that You
|
||||||
|
are, or that Your use of the Licensed Material is, connected
|
||||||
|
with, or sponsored, endorsed, or granted official status by,
|
||||||
|
the Licensor or others designated to receive attribution as
|
||||||
|
provided in Section 3(a)(1)(A)(i).
|
||||||
|
|
||||||
|
b. Other rights.
|
||||||
|
|
||||||
|
1. Moral rights, such as the right of integrity, are not
|
||||||
|
licensed under this Public License, nor are publicity,
|
||||||
|
privacy, and/or other similar personality rights; however, to
|
||||||
|
the extent possible, the Licensor waives and/or agrees not to
|
||||||
|
assert any such rights held by the Licensor to the limited
|
||||||
|
extent necessary to allow You to exercise the Licensed
|
||||||
|
Rights, but not otherwise.
|
||||||
|
|
||||||
|
2. Patent and trademark rights are not licensed under this
|
||||||
|
Public License.
|
||||||
|
|
||||||
|
3. To the extent possible, the Licensor waives any right to
|
||||||
|
collect royalties from You for the exercise of the Licensed
|
||||||
|
Rights, whether directly or through a collecting society
|
||||||
|
under any voluntary or waivable statutory or compulsory
|
||||||
|
licensing scheme. In all other cases the Licensor expressly
|
||||||
|
reserves any right to collect such royalties.
|
||||||
|
|
||||||
|
|
||||||
|
Section 3 -- License Conditions.
|
||||||
|
|
||||||
|
Your exercise of the Licensed Rights is expressly made subject to the
|
||||||
|
following conditions.
|
||||||
|
|
||||||
|
a. Attribution.
|
||||||
|
|
||||||
|
1. If You Share the Licensed Material (including in modified
|
||||||
|
form), You must:
|
||||||
|
|
||||||
|
a. retain the following if it is supplied by the Licensor
|
||||||
|
with the Licensed Material:
|
||||||
|
|
||||||
|
i. identification of the creator(s) of the Licensed
|
||||||
|
Material and any others designated to receive
|
||||||
|
attribution, in any reasonable manner requested by
|
||||||
|
the Licensor (including by pseudonym if
|
||||||
|
designated);
|
||||||
|
|
||||||
|
ii. a copyright notice;
|
||||||
|
|
||||||
|
iii. a notice that refers to this Public License;
|
||||||
|
|
||||||
|
iv. a notice that refers to the disclaimer of
|
||||||
|
warranties;
|
||||||
|
|
||||||
|
v. a URI or hyperlink to the Licensed Material to the
|
||||||
|
extent reasonably practicable;
|
||||||
|
|
||||||
|
b. indicate if You modified the Licensed Material and
|
||||||
|
retain an indication of any previous modifications; and
|
||||||
|
|
||||||
|
c. indicate the Licensed Material is licensed under this
|
||||||
|
Public License, and include the text of, or the URI or
|
||||||
|
hyperlink to, this Public License.
|
||||||
|
|
||||||
|
2. You may satisfy the conditions in Section 3(a)(1) in any
|
||||||
|
reasonable manner based on the medium, means, and context in
|
||||||
|
which You Share the Licensed Material. For example, it may be
|
||||||
|
reasonable to satisfy the conditions by providing a URI or
|
||||||
|
hyperlink to a resource that includes the required
|
||||||
|
information.
|
||||||
|
|
||||||
|
3. If requested by the Licensor, You must remove any of the
|
||||||
|
information required by Section 3(a)(1)(A) to the extent
|
||||||
|
reasonably practicable.
|
||||||
|
|
||||||
|
4. If You Share Adapted Material You produce, the Adapter's
|
||||||
|
License You apply must not prevent recipients of the Adapted
|
||||||
|
Material from complying with this Public License.
|
||||||
|
|
||||||
|
|
||||||
|
Section 4 -- Sui Generis Database Rights.
|
||||||
|
|
||||||
|
Where the Licensed Rights include Sui Generis Database Rights that
|
||||||
|
apply to Your use of the Licensed Material:
|
||||||
|
|
||||||
|
a. for the avoidance of doubt, Section 2(a)(1) grants You the right
|
||||||
|
to extract, reuse, reproduce, and Share all or a substantial
|
||||||
|
portion of the contents of the database;
|
||||||
|
|
||||||
|
b. if You include all or a substantial portion of the database
|
||||||
|
contents in a database in which You have Sui Generis Database
|
||||||
|
Rights, then the database in which You have Sui Generis Database
|
||||||
|
Rights (but not its individual contents) is Adapted Material; and
|
||||||
|
|
||||||
|
c. You must comply with the conditions in Section 3(a) if You Share
|
||||||
|
all or a substantial portion of the contents of the database.
|
||||||
|
|
||||||
|
For the avoidance of doubt, this Section 4 supplements and does not
|
||||||
|
replace Your obligations under this Public License where the Licensed
|
||||||
|
Rights include other Copyright and Similar Rights.
|
||||||
|
|
||||||
|
|
||||||
|
Section 5 -- Disclaimer of Warranties and Limitation of Liability.
|
||||||
|
|
||||||
|
a. UNLESS OTHERWISE SEPARATELY UNDERTAKEN BY THE LICENSOR, TO THE
|
||||||
|
EXTENT POSSIBLE, THE LICENSOR OFFERS THE LICENSED MATERIAL AS-IS
|
||||||
|
AND AS-AVAILABLE, AND MAKES NO REPRESENTATIONS OR WARRANTIES OF
|
||||||
|
ANY KIND CONCERNING THE LICENSED MATERIAL, WHETHER EXPRESS,
|
||||||
|
IMPLIED, STATUTORY, OR OTHER. THIS INCLUDES, WITHOUT LIMITATION,
|
||||||
|
WARRANTIES OF TITLE, MERCHANTABILITY, FITNESS FOR A PARTICULAR
|
||||||
|
PURPOSE, NON-INFRINGEMENT, ABSENCE OF LATENT OR OTHER DEFECTS,
|
||||||
|
ACCURACY, OR THE PRESENCE OR ABSENCE OF ERRORS, WHETHER OR NOT
|
||||||
|
KNOWN OR DISCOVERABLE. WHERE DISCLAIMERS OF WARRANTIES ARE NOT
|
||||||
|
ALLOWED IN FULL OR IN PART, THIS DISCLAIMER MAY NOT APPLY TO YOU.
|
||||||
|
|
||||||
|
b. TO THE EXTENT POSSIBLE, IN NO EVENT WILL THE LICENSOR BE LIABLE
|
||||||
|
TO YOU ON ANY LEGAL THEORY (INCLUDING, WITHOUT LIMITATION,
|
||||||
|
NEGLIGENCE) OR OTHERWISE FOR ANY DIRECT, SPECIAL, INDIRECT,
|
||||||
|
INCIDENTAL, CONSEQUENTIAL, PUNITIVE, EXEMPLARY, OR OTHER LOSSES,
|
||||||
|
COSTS, EXPENSES, OR DAMAGES ARISING OUT OF THIS PUBLIC LICENSE OR
|
||||||
|
USE OF THE LICENSED MATERIAL, EVEN IF THE LICENSOR HAS BEEN
|
||||||
|
ADVISED OF THE POSSIBILITY OF SUCH LOSSES, COSTS, EXPENSES, OR
|
||||||
|
DAMAGES. WHERE A LIMITATION OF LIABILITY IS NOT ALLOWED IN FULL OR
|
||||||
|
IN PART, THIS LIMITATION MAY NOT APPLY TO YOU.
|
||||||
|
|
||||||
|
c. The disclaimer of warranties and limitation of liability provided
|
||||||
|
above shall be interpreted in a manner that, to the extent
|
||||||
|
possible, most closely approximates an absolute disclaimer and
|
||||||
|
waiver of all liability.
|
||||||
|
|
||||||
|
|
||||||
|
Section 6 -- Term and Termination.
|
||||||
|
|
||||||
|
a. This Public License applies for the term of the Copyright and
|
||||||
|
Similar Rights licensed here. However, if You fail to comply with
|
||||||
|
this Public License, then Your rights under this Public License
|
||||||
|
terminate automatically.
|
||||||
|
|
||||||
|
b. Where Your right to use the Licensed Material has terminated under
|
||||||
|
Section 6(a), it reinstates:
|
||||||
|
|
||||||
|
1. automatically as of the date the violation is cured, provided
|
||||||
|
it is cured within 30 days of Your discovery of the
|
||||||
|
violation; or
|
||||||
|
|
||||||
|
2. upon express reinstatement by the Licensor.
|
||||||
|
|
||||||
|
For the avoidance of doubt, this Section 6(b) does not affect any
|
||||||
|
right the Licensor may have to seek remedies for Your violations
|
||||||
|
of this Public License.
|
||||||
|
|
||||||
|
c. For the avoidance of doubt, the Licensor may also offer the
|
||||||
|
Licensed Material under separate terms or conditions or stop
|
||||||
|
distributing the Licensed Material at any time; however, doing so
|
||||||
|
will not terminate this Public License.
|
||||||
|
|
||||||
|
d. Sections 1, 5, 6, 7, and 8 survive termination of this Public
|
||||||
|
License.
|
||||||
|
|
||||||
|
|
||||||
|
Section 7 -- Other Terms and Conditions.
|
||||||
|
|
||||||
|
a. The Licensor shall not be bound by any additional or different
|
||||||
|
terms or conditions communicated by You unless expressly agreed.
|
||||||
|
|
||||||
|
b. Any arrangements, understandings, or agreements regarding the
|
||||||
|
Licensed Material not stated herein are separate from and
|
||||||
|
independent of the terms and conditions of this Public License.
|
||||||
|
|
||||||
|
|
||||||
|
Section 8 -- Interpretation.
|
||||||
|
|
||||||
|
a. For the avoidance of doubt, this Public License does not, and
|
||||||
|
shall not be interpreted to, reduce, limit, restrict, or impose
|
||||||
|
conditions on any use of the Licensed Material that could lawfully
|
||||||
|
be made without permission under this Public License.
|
||||||
|
|
||||||
|
b. To the extent possible, if any provision of this Public License is
|
||||||
|
deemed unenforceable, it shall be automatically reformed to the
|
||||||
|
minimum extent necessary to make it enforceable. If the provision
|
||||||
|
cannot be reformed, it shall be severed from this Public License
|
||||||
|
without affecting the enforceability of the remaining terms and
|
||||||
|
conditions.
|
||||||
|
|
||||||
|
c. No term or condition of this Public License will be waived and no
|
||||||
|
failure to comply consented to unless expressly agreed to by the
|
||||||
|
Licensor.
|
||||||
|
|
||||||
|
d. Nothing in this Public License constitutes or may be interpreted
|
||||||
|
as a limitation upon, or waiver of, any privileges and immunities
|
||||||
|
that apply to the Licensor or You, including from the legal
|
||||||
|
processes of any jurisdiction or authority.
|
||||||
|
|
||||||
|
|
||||||
|
=======================================================================
|
||||||
|
|
||||||
|
Creative Commons is not a party to its public
|
||||||
|
licenses. Notwithstanding, Creative Commons may elect to apply one of
|
||||||
|
its public licenses to material it publishes and in those instances
|
||||||
|
will be considered the “Licensor.” The text of the Creative Commons
|
||||||
|
public licenses is dedicated to the public domain under the CC0 Public
|
||||||
|
Domain Dedication. Except for the limited purpose of indicating that
|
||||||
|
material is shared under a Creative Commons public license or as
|
||||||
|
otherwise permitted by the Creative Commons policies published at
|
||||||
|
creativecommons.org/policies, Creative Commons does not authorize the
|
||||||
|
use of the trademark "Creative Commons" or any other trademark or logo
|
||||||
|
of Creative Commons without its prior written consent including,
|
||||||
|
without limitation, in connection with any unauthorized modifications
|
||||||
|
to any of its public licenses or any other arrangements,
|
||||||
|
understandings, or agreements concerning use of licensed material. For
|
||||||
|
the avoidance of doubt, this paragraph does not form part of the
|
||||||
|
public licenses.
|
||||||
|
|
||||||
|
Creative Commons may be contacted at creativecommons.org.
|
||||||
|
|
||||||
@@ -1,40 +1,31 @@
|
|||||||
# pop-fem-audit
|
# 流行音樂中「女性力量」語彙的挪用與污染——以 Billboard Year-End Hot 100(2016-2025)為例的內容分析
|
||||||
|
|
||||||
流行音樂中「女性力量」語彙的挪用與污染——以 Billboard Year-End
|
這是研究論文《流行音樂中「女性力量」語彙的挪用與污染——以 Billboard Year-End Hot 100(2016-2025)為例的內容分析》的專案資料,包括論文本身、文件、紀錄、資料、工具程式、AI提示詞,等等。
|
||||||
Hot 100(2016–2025)為例的內容分析。
|
|
||||||
|
|
||||||
台灣女性學學會 2026 年會論文之研究資料與分析程式。
|
## 論文和摘要
|
||||||
|
|
||||||
## 目錄結構
|
研究論文全文和摘要,請參閱 paper/ 資料夾。
|
||||||
|
|
||||||
見 `docs/project-structure.md`。研究步驟規劃見
|
## 文件說明
|
||||||
`docs/research-plan.md`;方法細節見 `docs/methodology.md`;
|
|
||||||
人工編碼手冊見 `docs/codebook.md`;決策日誌見
|
|
||||||
`docs/decision-log.md`。
|
|
||||||
|
|
||||||
## 重現方式
|
研究方法、記錄等文件,請參閱 docs/ 資料夾。
|
||||||
|
|
||||||
1. 準備 Python 3.14+ 環境,安裝分析管線套件:
|
## 輔助工具程式
|
||||||
`pip install -e tools/`。
|
|
||||||
2. 自 `tools/.env.example` 建立 `tools/.env`,寫入
|
|
||||||
Anthropic API 金鑰。
|
|
||||||
3. 依 `docs/research-plan.md` 的階段順序,於 `tools/` 目錄下
|
|
||||||
執行子命令,必要輸入以位置引數、選擇性輸入以選項給定(如
|
|
||||||
`pop-fem-audit-tools build-db
|
|
||||||
../data/source/yearend_hot100_2016_2025.csv
|
|
||||||
../data/derived
|
|
||||||
--lyrics-dir ../data/captures/lyrics
|
|
||||||
--wikidata-csv ../data/captures/artists-wikidata.csv`)。
|
|
||||||
LLM 步驟使用 `claude-sonnet-4-6`、temperature=0、
|
|
||||||
thinking 關閉;輸出可逐項比對的步驟獨立執行三次,由
|
|
||||||
程式取三票多數決定案(「三次執行+多數決」協定),
|
|
||||||
自由生成步驟兩次執行進池。
|
|
||||||
4. 每次執行的完整紀錄(定義檔快照、原始輸出、參數)存於
|
|
||||||
`runs/`,可逐筆稽核。論文引用的最終資料表在 `results/`。
|
|
||||||
|
|
||||||
注意:歌詞受版權保護,`data/captures/lyrics/` 不隨 repo 發布,須自行
|
輔助工具程式,請參閱 tools/ 資料夾。
|
||||||
以 `tools/` 中的抓取程式重建。
|
|
||||||
|
|
||||||
## 授權
|
## 授權
|
||||||
|
|
||||||
(待定)
|
版權所有 © 2026 楊士青。
|
||||||
|
|
||||||
|
除另有註明外,本專案之內容均依「CC 姓名標示 4.0 國際」授權條款提供。
|
||||||
|
|
||||||
|
排除物件:
|
||||||
|
1. 本專案引用之第三方歌詞,相應權利歸各該權利人所有;
|
||||||
|
2. tools/ 資料夾內之工具程式,依該工具程式授權條款提供。
|
||||||
|
|
||||||
|
## 作者
|
||||||
|
|
||||||
|
楊士青<br>
|
||||||
|
imacat@mail.imacat.idv.tw<br>
|
||||||
|
2026/8/18<br>
|
||||||
|
|||||||
@@ -1,7 +1,17 @@
|
|||||||
# Project Conventions
|
# Project Conventions
|
||||||
|
|
||||||
|
The standing working rules of this project. Formerly the
|
||||||
|
project `CLAUDE.md`; moved here so that Claude Code subagents
|
||||||
|
do not inherit it into their context (blind-reading agents
|
||||||
|
must not see it). A main session working on this project
|
||||||
|
reads this file before touching the pipeline, the data, or
|
||||||
|
the documents.
|
||||||
|
|
||||||
## Analysis pipeline
|
## Analysis pipeline
|
||||||
|
|
||||||
|
The pipeline has run to completion; these conventions govern
|
||||||
|
any rerun or extension.
|
||||||
|
|
||||||
- LLM analysis runs via Python scripts calling the Anthropic
|
- LLM analysis runs via Python scripts calling the Anthropic
|
||||||
Messages API, Batch API where possible. Steps 1 and 3 run
|
Messages API, Batch API where possible. Steps 1 and 3 run
|
||||||
on `claude-sonnet-4-6` with `temperature=0` and thinking
|
on `claude-sonnet-4-6` with `temperature=0` and thinking
|
||||||
@@ -10,13 +20,13 @@
|
|||||||
variance by the majority vote, step 5 by consolidating the
|
variance by the majority vote, step 5 by consolidating the
|
||||||
three readings.
|
three readings.
|
||||||
- Prompt definition files live in
|
- Prompt definition files live in
|
||||||
`prompts/<step>-<substep>-<task>.md` (e.g. 01-tag.md; no
|
`prompts/<step><substep>-<task>.md` (e.g. 1-tag.md,
|
||||||
version suffix -- versions live in git history) and are
|
5a-read.md; substeps are lettered, matching the step
|
||||||
passed verbatim as the system prompt. The number names a
|
numbering of the paper; no version suffix -- versions live
|
||||||
step of the research procedure, not the file: the
|
in git history) and are passed verbatim as the system
|
||||||
deterministic vocabulary step (step 2) has no definition
|
prompt. The number names a step of the research
|
||||||
file yet holds its own number. Zero padding is for
|
procedure, not the file: the deterministic vocabulary step
|
||||||
sorting only -- prose says "step 1", "step 3".
|
(step 2) has no definition file yet holds its own number.
|
||||||
- Itemwise LLM judgments (per-song coding in step 3,
|
- Itemwise LLM judgments (per-song coding in step 3,
|
||||||
per-keyword group selection in step 4) run the same
|
per-keyword group selection in step 4) run the same
|
||||||
definition file three times, independently, over the same
|
definition file three times, independently, over the same
|
||||||
@@ -64,5 +74,5 @@
|
|||||||
- `results/` holds the final tallied tables (what the paper
|
- `results/` holds the final tallied tables (what the paper
|
||||||
cites); `runs/` holds raw audit records. The paper cites
|
cites); `runs/` holds raw audit records. The paper cites
|
||||||
`results/` only.
|
`results/` only.
|
||||||
- Any change to a definition file, the codebook, or the plan is
|
- Any change to a definition file or the plan is recorded in
|
||||||
recorded in `docs/decision-log.md` with date and reason.
|
`docs/decision-log.md` with date and reason.
|
||||||
+63
-17
@@ -99,8 +99,8 @@
|
|||||||
|
|
||||||
## 2026-08-02
|
## 2026-08-02
|
||||||
|
|
||||||
- **`import-lyrics` 不設為子命令,pilot 歌詞改以私人腳本
|
- **`import-lyrics` 不設為子命令,pilot 歌詞改以 `excludes/`
|
||||||
匯入**。理由:pilot 捕捉檔不隨論文發布,子命令
|
私人腳本匯入**。理由:pilot 捕捉檔不隨論文發布,子命令
|
||||||
形式會在發布的 CLI 裡留下讀者無法執行的死命令——要交待的
|
形式會在發布的 CLI 裡留下讀者無法執行的死命令——要交待的
|
||||||
是「沿用 pilot 捕捉」的事實(記於 lyrics-provenance.csv 與
|
是「沿用 pilot 捕捉」的事實(記於 lyrics-provenance.csv 與
|
||||||
論文方法節),不是工具本身;工具移出專案,發布管線即
|
論文方法節),不是工具本身;工具移出專案,發布管線即
|
||||||
@@ -178,8 +178,8 @@
|
|||||||
來源公開可稽核,快照乾淨、overrides 縮小;論文方法節揭露
|
來源公開可稽核,快照乾淨、overrides 縮小;論文方法節揭露
|
||||||
「捕捉前經研究者查證補完」。性別以公開自我認同為準,
|
「捕捉前經研究者查證補完」。性別以公開自我認同為準,
|
||||||
推測不確定且無佐證者列疑慮清單;查證屬資料策展而非
|
推測不確定且無佐證者列疑慮清單;查證屬資料策展而非
|
||||||
分析,以 Claude Code 輔助、不走分析 API;QS 批次以
|
分析,以 Claude Code 輔助、不走分析 API;QS 批次存
|
||||||
私人工作檔留存,不隨論文發布。
|
`excludes/`(私人工作檔,不隨論文發布)。
|
||||||
|
|
||||||
## 2026-08-04
|
## 2026-08-04
|
||||||
|
|
||||||
@@ -227,7 +227,7 @@
|
|||||||
- **Markdown 檔名一律以 dash 連接**(decision-log.md、
|
- **Markdown 檔名一律以 dash 連接**(decision-log.md、
|
||||||
research-plan.md、project-structure.md 等;與 data/ 層
|
research-plan.md、project-structure.md 等;與 data/ 層
|
||||||
CSV 檔名慣例一致),定義檔命名慣例同步改為
|
CSV 檔名慣例一致),定義檔命名慣例同步改為
|
||||||
`prompts/<task>-v<N>.md`(如 screen-v1.md),版本庫外的
|
`prompts/<task>-v<N>.md`(如 screen-v1.md),`excludes/`
|
||||||
私人工作檔一併改名;`conference_abstract.md` 改名並搬入
|
私人工作檔一併改名;`conference_abstract.md` 改名並搬入
|
||||||
`paper/`——它是本次年會實際送出的摘要,與全文同屬投稿
|
`paper/`——它是本次年會實際送出的摘要,與全文同屬投稿
|
||||||
血脈,不是 `docs/` 的內部工作文件。全 repo 指涉同步更新。
|
血脈,不是 `docs/` 的內部工作文件。全 repo 指涉同步更新。
|
||||||
@@ -583,8 +583,8 @@
|
|||||||
Billboard 署名 Pinkfong 為品牌(Wikidata Q55735607,型態
|
Billboard 署名 Pinkfong 為品牌(Wikidata Q55735607,型態
|
||||||
brand),非演唱者;其唯一上榜曲 Baby Shark(2019#75)
|
brand),非演唱者;其唯一上榜曲 Baby Shark(2019#75)
|
||||||
的實際演唱者為 Hope Segoine(KTVB 2019-03-06 報導、
|
的實際演唱者為 Hope Segoine(KTVB 2019-03-06 報導、
|
||||||
Songfacts、經紀簡介,出處詳私人留存的
|
Songfacts、經紀簡介,出處詳
|
||||||
QuickStatements 草稿)。藉既有的
|
`excludes/quickstatements/hope-segoine.md`)。藉既有的
|
||||||
署名正規化機制(`ArtistImporter.CANONICAL_ARTIST_NAMES`)
|
署名正規化機制(`ArtistImporter.CANONICAL_ARTIST_NAMES`)
|
||||||
指認:歌曲的署名字串維持榜單所印的 "Pinkfong",解析出
|
指認:歌曲的署名字串維持榜單所印的 "Pinkfong",解析出
|
||||||
的演出者實體為 Hope Segoine。品牌無性別可言,人有;
|
的演出者實體為 Hope Segoine。品牌無性別可言,人有;
|
||||||
@@ -637,9 +637,10 @@
|
|||||||
(sentence-transformers/all-mpnet-base-v2, revision
|
(sentence-transformers/all-mpnet-base-v2, revision
|
||||||
e8c3b32e)對 `runs/02-cluster/source-keywords.txt` 的
|
e8c3b32e)對 `runs/02-cluster/source-keywords.txt` 的
|
||||||
5,999 個關鍵字重新計算。組內一致性定義為組內成員兩兩
|
5,999 個關鍵字重新計算。組內一致性定義為組內成員兩兩
|
||||||
餘弦相似度之平均。k=50 直接取自私人留存的分群結果;
|
餘弦相似度之平均。k=50 直接取自留存的分群結果
|
||||||
k=30 以同參數重跑(確定性演算法),重算所得最大組恰為
|
(`excludes/k50/02-cluster/groups.csv`);k=30 以同參數
|
||||||
491 詞,與當時粗算紀錄吻合,佐證重算與原實驗一致。數表:
|
重跑(確定性演算法),重算所得最大組恰為 491 詞,與
|
||||||
|
當時粗算紀錄吻合,佐證重算與原實驗一致。數表:
|
||||||
|
|
||||||
| | k=30 | k=50 | k=100 |
|
| | k=30 | k=50 | k=100 |
|
||||||
|---|---|---|---|
|
|---|---|---|---|
|
||||||
@@ -702,8 +703,8 @@
|
|||||||
力量語意即入選,women-power 群 11 碼,並產詞彙表外
|
力量語意即入選,women-power 群 11 碼,並產詞彙表外
|
||||||
幻覺碼一筆),claude-fable-5 讀為詞彙化概念(要求女性
|
幻覺碼一筆),claude-fable-5 讀為詞彙化概念(要求女性
|
||||||
標記,women-power 群 2 碼),與草稿盲選及深度閱讀輔助
|
標記,women-power 群 2 碼),與草稿盲選及深度閱讀輔助
|
||||||
判讀同讀法;sonnet 對照執行歸檔另行私人備份,不入
|
判讀同讀法;sonnet 對照執行歸檔備份於
|
||||||
版本庫,支出留帳。claude-fable-5 不
|
`excludes/experiments/`,支出留帳。claude-fable-5 不
|
||||||
受理 temperature 與 thinking 參數(均不送出),無法釘
|
受理 temperature 與 thinking 參數(均不送出),無法釘
|
||||||
temperature=0;執行間變異實測存在(run1/run2 於陽剛、
|
temperature=0;執行間變異實測存在(run1/run2 於陽剛、
|
||||||
脆弱邊緣碼分歧),由三票多數決吸收;詞彙表外輸出項
|
脆弱邊緣碼分歧),由三票多數決吸收;詞彙表外輸出項
|
||||||
@@ -750,9 +751,9 @@
|
|||||||
編碼、票數),納入重建摘要與清空範圍,供群層次查詢;
|
編碼、票數),納入重建摘要與清空範圍,供群層次查詢;
|
||||||
重建命令自此帶 `--groups results/groups.csv`。
|
重建命令自此帶 `--groups results/groups.csv`。
|
||||||
|
|
||||||
- **Fable 5 輔助判讀實驗總錄(非正式編碼;結果檔私人
|
- **Fable 5 輔助判讀實驗總錄(非正式編碼;結果檔於
|
||||||
留存、不入版本庫,本條為其版本庫內的程序錨點)**:
|
gitignored 之 `excludes/`,本條為其版本庫內的程序
|
||||||
2026-08-08 起以 Claude Code subagent(模型
|
錨點)**:2026-08-08 起以 Claude Code subagent(模型
|
||||||
Fable 5)進行七項深度閱讀實驗。共同程序:每首歌一個
|
Fable 5)進行七項深度閱讀實驗。共同程序:每首歌一個
|
||||||
獨立會話、提示最少化且逐字統一、判讀者互不知情、不經
|
獨立會話、提示最少化且逐字統一、判讀者互不知情、不經
|
||||||
三票制;定位為研究者的輔助判讀(初篩),不改動任何
|
三票制;定位為研究者的輔助判讀(初篩),不改動任何
|
||||||
@@ -782,7 +783,7 @@
|
|||||||
`male-mixed-wp-fe-feminist-reading.md` 與同名 .ods)。
|
`male-mixed-wp-fe-feminist-reading.md` 與同名 .ods)。
|
||||||
此系列即論文方法節「最後編碼的結果,再由研究者與 LLM
|
此系列即論文方法節「最後編碼的結果,再由研究者與 LLM
|
||||||
(Claude Code)輔助判讀」之所指;結果檔含大量歌詞引文,
|
(Claude Code)輔助判讀」之所指;結果檔含大量歌詞引文,
|
||||||
依著作權紀律私人留置,不入版本庫。
|
依著作權紀律留置 `excludes/`,不入版本庫。
|
||||||
|
|
||||||
- **步驟 5:女性主義問題之質性深讀(設計定案)**:論文
|
- **步驟 5:女性主義問題之質性深讀(設計定案)**:論文
|
||||||
題目「挪用與污染」需要框架層的系統性證據;47 首
|
題目「挪用與污染」需要框架層的系統性證據;47 首
|
||||||
@@ -898,4 +899,49 @@
|
|||||||
performer_gender 手工修正)、四份 results 定案表、時程
|
performer_gender 手工修正)、四份 results 定案表、時程
|
||||||
展延一至二日與剩餘撰寫工作。仍然有效的原則(先導不
|
展延一至二日與剩餘撰寫工作。仍然有效的原則(先導不
|
||||||
比較、提示只定格式、歌手背景防火牆、commit 判準、
|
比較、提示只定格式、歌手背景防火牆、commit 判準、
|
||||||
版權規則)保留。
|
版權規則、positionality)保留。
|
||||||
|
|
||||||
|
## 2026-08-17
|
||||||
|
|
||||||
|
- **先導研究的地位另立常設紀錄 `docs/pilot-study.md`**:論文
|
||||||
|
交件後重讀投稿摘要,認清先導研究係由當時協作的 Claude Code
|
||||||
|
設計並執行,研究者檢視的是研究問題與答案,未逐步檢視中間
|
||||||
|
過程;摘要中的具體數字未經稽核,亦無歸檔可回溯。此地位
|
||||||
|
先前僅零星散見於本日誌各條(歌詞沿用、women-power 來歷
|
||||||
|
考據、粒度繼承、簿記容量),投稿摘要本身則只是產物,不足以
|
||||||
|
記述過程。裁定:另立 `docs/pilot-study.md`,完整記述先導
|
||||||
|
研究的執行方式、沿用與棄用的成果,以及它如何促成正式研究
|
||||||
|
的可稽核設計;投稿摘要不於樹中另存副本,其內容留在 git
|
||||||
|
歷史(`paper/abstract.md` 的前身)。
|
||||||
|
|
||||||
|
## 2026-08-18
|
||||||
|
|
||||||
|
- **工序編號改與論文正文一致**:論文正文將五個步驟的細分
|
||||||
|
工序以字母標示(表三「步驟3a」、表四「步驟5d」),repo
|
||||||
|
的檔名與歸檔目錄卻沿用 `05-01`、`05-02` 這種數字次步,
|
||||||
|
兩邊對不上。裁定:改為 `1-tag`、`2-cluster`、`3a-code`、
|
||||||
|
`4-group`、`5a-read`、`5b-consolidate`、`5c-synthesize`、
|
||||||
|
`5d-annotate`(`prompts/` 與 `runs/` 同步,以 `git mv`
|
||||||
|
保留歷史),補零一併取消。`runs/*/meta.json`
|
||||||
|
內記的 `prompt_path` 不追改——它是執行當下的實況記錄;
|
||||||
|
本日誌的既有條目同理,維持當時的名稱。`run-costs.md` 則
|
||||||
|
相反:它的「步驟」欄是查閱歸檔的索引,故一律換為現行
|
||||||
|
編號(`01-01-01-tag`→`1-tag`、`03-01-code` 與
|
||||||
|
`03-code`→`3a-code`),並新增末欄「原步驟名」保留執行
|
||||||
|
當時的名稱——索引要指得到現在的歸檔,紀錄要留得住當時
|
||||||
|
的形狀,兩者以兩欄並存解決。已刪除的工序(步驟 2 的
|
||||||
|
LLM 合併嘗試 `01-02-01-merge`、步驟 3 的仲裁
|
||||||
|
`03-02-arbitration`)其歸檔已不存在,無現行編號可指,
|
||||||
|
不另發新名,步驟欄逕標「已廢棄」,原名存於末欄。
|
||||||
|
|
||||||
|
- **CLAUDE.md 移為 `docs/conventions.md`**:專案 CLAUDE.md 會
|
||||||
|
全文自動注入每一個 Claude Code subagent 的 context(2026-07-30
|
||||||
|
實測條),步驟 6 的 883 個盲判 agent 因此都看到了工作規範。
|
||||||
|
規範本身無研究語意(無 women-power、無假說),污染有限,但
|
||||||
|
「防脈絡污染」的原則應貫徹到自家工具鏈:docs/ 底下的檔案
|
||||||
|
不會自動注入,搬移後 subagent 除非主動翻閱否則不會看到,
|
||||||
|
盲判 agent 的定義檔並明令不得查閱專案材料。代價是主會話
|
||||||
|
也不再自動載入,須自行先讀 `docs/conventions.md`。內容
|
||||||
|
順帶微調:註明管線已完成、規範適用於重跑;刪去已不存在
|
||||||
|
的 codebook 一詞。進論文的數據一律走 Anthropic API(無
|
||||||
|
檔案系統,機制上不可能讀到),此層防線不受本次搬移影響。
|
||||||
|
|||||||
+23
-23
@@ -98,7 +98,7 @@
|
|||||||
## 步驟 3 編碼的三次執行與多數決
|
## 步驟 3 編碼的三次執行與多數決
|
||||||
|
|
||||||
- **三次執行**:同一份定義檔、同一份輸入檔,獨立執行
|
- **三次執行**:同一份定義檔、同一份輸入檔,獨立執行
|
||||||
三次,三份歸檔並列(`runs/03-code/run1`、`run2`、
|
三次,三份歸檔並列(`runs/3a-code/run1`、`run2`、
|
||||||
`run3`),彼此無先後主從之別。
|
`run3`),彼此無先後主從之別。
|
||||||
- **多數決**:一首歌的一個標籤,三次執行中至少兩次標出
|
- **多數決**:一首歌的一個標籤,三次執行中至少兩次標出
|
||||||
即收入定案編碼。三票不平手,裁決規則因此無例外條款,
|
即收入定案編碼。三票不平手,裁決規則因此無例外條款,
|
||||||
@@ -125,13 +125,13 @@
|
|||||||
由 LLM 依編碼名的字面語意判斷。
|
由 LLM 依編碼名的字面語意判斷。
|
||||||
- **任務**:每筆輸入為一個群名加 101 個編碼的字母序
|
- **任務**:每筆輸入為一個群名加 101 個編碼的字母序
|
||||||
清單,輸出為入選編碼的單層 JSON 陣列;定義檔
|
清單,輸出為入選編碼的單層 JSON 陣列;定義檔
|
||||||
`prompts/04-group.md` 只定格式,不含任何群的語意定義。
|
`prompts/4-group.md` 只定格式,不含任何群的語意定義。
|
||||||
- **模型**:`claude-fable-5`(步驟 1、3 為
|
- **模型**:`claude-fable-5`(步驟 1、3 為
|
||||||
`claude-sonnet-4-6`)。該模型不受理 `temperature` 與
|
`claude-sonnet-4-6`)。該模型不受理 `temperature` 與
|
||||||
`thinking` 參數,兩者均不送出;取樣變異由多數決吸收。
|
`thinking` 參數,兩者均不送出;取樣變異由多數決吸收。
|
||||||
模型裁定的理由與對照實驗見決策日誌。
|
模型裁定的理由與對照實驗見決策日誌。
|
||||||
- **三次執行**:同一份定義檔、同一份輸入檔,獨立執行
|
- **三次執行**:同一份定義檔、同一份輸入檔,獨立執行
|
||||||
三次,歸檔並列(`runs/04-group/run1`、`run2`、`run3`)。
|
三次,歸檔並列(`runs/4-group/run1`、`run2`、`run3`)。
|
||||||
- **多數決**:一個(群,編碼)配對,三次執行中至少兩次
|
- **多數決**:一個(群,編碼)配對,三次執行中至少兩次
|
||||||
入選即屬該群;不在 101 碼詞彙表內的輸出項無效,每筆
|
入選即屬該群;不在 101 碼詞彙表內的輸出項無效,每筆
|
||||||
丟棄印於標準錯誤。計票由確定性子命令 `tally-groups`
|
丟棄印於標準錯誤。計票由確定性子命令 `tally-groups`
|
||||||
@@ -146,7 +146,7 @@
|
|||||||
|
|
||||||
## 步驟 5 女性主義問題之質性深讀
|
## 步驟 5 女性主義問題之質性深讀
|
||||||
|
|
||||||
本步驟之 5-1 至 5-3 為**質性閱讀,非編碼**:輸出為自由
|
本步驟之 5a 至 5c 為**質性閱讀,非編碼**:輸出為自由
|
||||||
文字的問題閱讀報告,無可逐項機械比對的單位,故不適用
|
文字的問題閱讀報告,無可逐項機械比對的單位,故不適用
|
||||||
三票多數決與仲裁;
|
三票多數決與仲裁;
|
||||||
三次獨立閱讀為分析者三角檢核,逐首整合為整合而非裁決,
|
三次獨立閱讀為分析者三角檢核,逐首整合為整合而非裁決,
|
||||||
@@ -155,34 +155,34 @@
|
|||||||
|
|
||||||
- **對象**:定案編碼含 `women-power` 或
|
- **對象**:定案編碼含 `women-power` 或
|
||||||
`female-empowerment` 的 145 首歌。
|
`female-empowerment` 的 145 首歌。
|
||||||
- **5-1 逐首閱讀**:每筆輸入為一首歌的完整歌詞逐字全文,
|
- **5a 逐首閱讀**:每筆輸入為一首歌的完整歌詞逐字全文,
|
||||||
不含歌名與演唱者(盲讀);定義檔
|
不含歌名與演唱者(盲讀);定義檔
|
||||||
`prompts/05-01-read.md`。同一份定義檔、同一份輸入檔,
|
`prompts/5a-read.md`。同一份定義檔、同一份輸入檔,
|
||||||
獨立執行三次,歸檔並列(`runs/05-01-read/run1`、
|
獨立執行三次,歸檔並列(`runs/5a-read/run1`、
|
||||||
`run2`、`run3`)。
|
`run2`、`run3`)。
|
||||||
- **5-2 逐首整合**:每筆輸入為該首歌的三份閱讀報告
|
- **5b 逐首整合**:每筆輸入為該首歌的三份閱讀報告
|
||||||
(不含歌詞);以問題機制為單位保守合併,標收斂註記
|
(不含歌詞);以問題機制為單位保守合併,標收斂註記
|
||||||
((3/3)、(2/3)),主清單僅列兩讀以上提出者,單讀發現
|
((3/3)、(2/3)),主清單僅列兩讀以上提出者,單讀發現
|
||||||
以一行存目;定義檔 `prompts/05-02-consolidate.md`,
|
以一行存目;定義檔 `prompts/5b-consolidate.md`,
|
||||||
執行一次,歸檔 `runs/05-02-consolidate/run1`。
|
執行一次,歸檔 `runs/5b-consolidate/run1`。
|
||||||
- **5-3 樣態統整**:單筆輸入為一批整合報告;歸納問題
|
- **5c 樣態統整**:單筆輸入為一批整合報告;歸納問題
|
||||||
**樣態**——問題呈現與運作的重複形態,非問題分類,
|
**樣態**——問題呈現與運作的重複形態,非問題分類,
|
||||||
代表引句僅取自主清單;定義檔
|
代表引句僅取自主清單;定義檔
|
||||||
`prompts/05-03-synthesize.md`。四種輸入範圍各執行
|
`prompts/5c-synthesize.md`。四種輸入範圍各執行
|
||||||
一次:全 145 首之基底統整(歸檔
|
一次:全 145 首之基底統整(歸檔
|
||||||
`runs/05-03-synthesize/run1`),及依 `performer_gender`
|
`runs/5c-synthesize/run1`),及依 `performer_gender`
|
||||||
(演唱聲音之性別)切分之三個發話脈絡統整——男聲
|
(演唱聲音之性別)切分之三個發話脈絡統整——男聲
|
||||||
(male,歸檔 `run2`)、女聲(female,歸檔 `run3`)、
|
(male,歸檔 `run2`)、女聲(female,歸檔 `run3`)、
|
||||||
混合(mixed,歸檔 `run4`);基底看橫貫各脈絡之樣態,
|
混合(mixed,歸檔 `run4`);基底看橫貫各脈絡之樣態,
|
||||||
分組看各權力脈絡下之樣態。genderfluid 與 non-binary
|
分組看各權力脈絡下之樣態。genderfluid 與 non-binary
|
||||||
共 3 首不設群——樣本數不支持歸納——僅入基底統整,
|
共 3 首不設群——樣本數不支持歸納——僅入基底統整,
|
||||||
由研究者以個案閱讀。同一輸入不重複執行:自由歸納之
|
由研究者以個案閱讀。同一輸入不重複執行:自由歸納之
|
||||||
產出無機械合併可言,其變異由 5-1 三讀、5-2 整合與
|
產出無機械合併可言,其變異由 5a 三讀、5b 整合與
|
||||||
草稿地位承接,四份草稿互為對照,由研究者終審裁決。
|
草稿地位承接,四份草稿互為對照,由研究者終審裁決。
|
||||||
- **5-4 樣態標註**:以 (歌, 樣態) 對為可逐項機械比對之
|
- **5d 樣態標註**:以 (歌, 樣態) 對為可逐項機械比對之
|
||||||
單位,回歸「三次執行+多數決」協定。對象為「有問題」
|
單位,回歸「三次執行+多數決」協定。對象為「有問題」
|
||||||
的歌——三讀中至多一個「無」(多數決精神;恰兩「無」
|
的歌——三讀中至多一個「無」(多數決精神;恰兩「無」
|
||||||
者其整合報告主清單必為空,與 5-2 主清單規則自洽),
|
者其整合報告主清單必為空,與 5b 主清單規則自洽),
|
||||||
計 111 首(男聲 12、女聲 69、混合 29、genderfluid 1)。
|
計 111 首(男聲 12、女聲 69、混合 29、genderfluid 1)。
|
||||||
樣態表為三份分組統整草稿原文:男聲 13 條(M1–M13)、
|
樣態表為三份分組統整草稿原文:男聲 13 條(M1–M13)、
|
||||||
女聲 14 條(F1–F14)、混合 16 條(X1–X16);基底 15 條
|
女聲 14 條(F1–F14)、混合 16 條(X1–X16);基底 15 條
|
||||||
@@ -195,17 +195,17 @@
|
|||||||
位置無法先驗決定),其歸屬引用維持個案地位。每筆
|
位置無法先驗決定),其歸屬引用維持個案地位。每筆
|
||||||
輸入為一首歌之整合報告(「僅單獨提及」行於組裝時
|
輸入為一首歌之整合報告(「僅單獨提及」行於組裝時
|
||||||
剝除,標註僅依主清單)與該首適用之樣態表;定義檔
|
剝除,標註僅依主清單)與該首適用之樣態表;定義檔
|
||||||
`prompts/05-04-annotate.md`,獨立執行三次
|
`prompts/5d-annotate.md`,獨立執行三次
|
||||||
(`runs/05-04-annotate/run1`~`run3`),(歌, 樣態) 對
|
(`runs/5d-annotate/run1`~`run3`),(歌, 樣態) 對
|
||||||
得兩票以上者定案。
|
得兩票以上者定案。
|
||||||
- **模型**:`claude-fable-5`(與先導深讀同儀器;
|
- **模型**:`claude-fable-5`(與先導深讀同儀器;
|
||||||
`temperature` 與 `thinking` 參數不適用,均不送出)。
|
`temperature` 與 `thinking` 參數不適用,均不送出)。
|
||||||
- **輸入組裝**:確定性行內腳本。5-1:145 首依歌曲 ID
|
- **輸入組裝**:確定性行內腳本。5a:145 首依歌曲 ID
|
||||||
升序,`content` 為歌詞逐字全文;5-2:每筆
|
升序,`content` 為歌詞逐字全文;5b:每筆
|
||||||
`{"reports": [run1 輸出, run2 輸出, run3 輸出]}`;
|
`{"reports": [run1 輸出, run2 輸出, run3 輸出]}`;
|
||||||
5-3:單筆以 `song-<ID>` 為鍵、整合報告為值之 JSON
|
5c:單筆以 `song-<ID>` 為鍵、整合報告為值之 JSON
|
||||||
物件,鍵集合為該次統整之範圍(基底為全 145 首,
|
物件,鍵集合為該次統整之範圍(基底為全 145 首,
|
||||||
分組依工作庫 `performer_gender` 切分);5-4:每筆
|
分組依工作庫 `performer_gender` 切分);5d:每筆
|
||||||
`{"report": 主清單, "patterns": [{"id", "name",
|
`{"report": 主清單, "patterns": [{"id", "name",
|
||||||
"description"}]}`,樣態條目自分組統整草稿機械切出,
|
"description"}]}`,樣態條目自分組統整草稿機械切出,
|
||||||
代表引句不隨附——引句出自特定歌曲,判該曲時形同
|
代表引句不隨附——引句出自特定歌曲,判該曲時形同
|
||||||
@@ -263,7 +263,7 @@
|
|||||||
在資料中找不到對應者,即中止;校定的判準記於
|
在資料中找不到對應者,即中止;校定的判準記於
|
||||||
`decision-log.md`。
|
`decision-log.md`。
|
||||||
- **合法碼清單**:純文字、一行一個碼,自詞彙表產出:
|
- **合法碼清單**:純文字、一行一個碼,自詞彙表產出:
|
||||||
`{ cat runs/02-cluster/result-keywords.txt; echo
|
`{ cat runs/2-cluster/result-keywords.txt; echo
|
||||||
women-power; } | sort`。
|
women-power; } | sort`。
|
||||||
- **定案表**:`results/codings.csv`,欄位 `Song`、
|
- **定案表**:`results/codings.csv`,欄位 `Song`、
|
||||||
`Artist Credit`、`Keyword`、`Quote`,一列一個標籤。歌名
|
`Artist Credit`、`Keyword`、`Quote`,一列一個標籤。歌名
|
||||||
|
|||||||
@@ -1,11 +1,11 @@
|
|||||||
# LLM 輸出的契約查核
|
# LLM 輸出的契約查核
|
||||||
|
|
||||||
(2026-08-06 量測。對象為步驟 3 的三份執行歸檔
|
(2026-08-06 量測。對象為步驟 3 的三份執行歸檔
|
||||||
`runs/03-code/run1`–`run3`,共 883 首歌、44,149 筆標籤
|
`runs/3a-code/run1`–`run3`,共 883 首歌、44,149 筆標籤
|
||||||
指派、44,146 句引述。歌詞以模型實際看到的那一份為準,即
|
指派、44,146 句引述。歌詞以模型實際看到的那一份為準,即
|
||||||
`tools/instance/llm-input-code.jsonl`。)
|
`tools/instance/llm-input-code.jsonl`。)
|
||||||
|
|
||||||
定義檔 `prompts/03-code.md` 對輸出下了四條明確要求:只用
|
定義檔 `prompts/3a-code.md` 對輸出下了四條明確要求:只用
|
||||||
給定的碼且拼寫照給、每個標出的碼恰引一行歌詞、引述逐字、
|
給定的碼且拼寫照給、每個標出的碼恰引一行歌詞、引述逐字、
|
||||||
輸出為合法 JSON 且無其他文字。本檔記錄這四條各自被遵守到
|
輸出為合法 JSON 且無其他文字。本檔記錄這四條各自被遵守到
|
||||||
什麼程度。
|
什麼程度。
|
||||||
|
|||||||
@@ -0,0 +1,107 @@
|
|||||||
|
# 先導研究的來歷與地位
|
||||||
|
|
||||||
|
(2026-08-17 認清並記錄。本檔記述正式研究之前的先導研究:
|
||||||
|
它做了什麼、研究者檢視到哪一層、哪些成果被沿用、哪些被
|
||||||
|
棄用,以及它如何促成正式研究的設計。散見於
|
||||||
|
`decision-log.md` 的相關條目在文中逐一指出。)
|
||||||
|
|
||||||
|
## 一、先導研究是什麼
|
||||||
|
|
||||||
|
正式研究之前,研究者曾以 Claude Code 對 Billboard Year-End
|
||||||
|
Hot 100(2018–2025、684 首)的歌詞做過一輪探索性分析,檢視
|
||||||
|
「女性力量」語彙的使用狀況,並附帶分析 pussy 一詞作為女性
|
||||||
|
代稱的修辭。2026 年 8 月投出的研討會摘要即根據該輪分析撰寫。
|
||||||
|
|
||||||
|
## 二、它實際的執行方式
|
||||||
|
|
||||||
|
**研究問題與方向由研究者提出**:女性力量語彙被父權框架取用、
|
||||||
|
pussy 作為代稱的跨性別使用,都是研究者長期關注的問題。
|
||||||
|
|
||||||
|
**設計與執行由 Claude Code 進行**:分析管線、分類架構(genuine
|
||||||
|
/peripheral/fake 三分類)、五種「假女性力量」類型、以及
|
||||||
|
frame-aware 提示修正實驗,均由當時協作的 Claude Code 設計並
|
||||||
|
執行。
|
||||||
|
|
||||||
|
**研究者檢視的是問題與答案,不是中間過程**:研究者確認研究
|
||||||
|
問題的方向正確、答案值得追究,即據以投稿;分類類目如何產生、
|
||||||
|
提示詞如何下、數字如何算出,並未逐步檢視。投稿摘要中的具體
|
||||||
|
數字(如「44% 屬於假女性力量」、62 首的三分類、53 首的 pussy
|
||||||
|
修辭分類、假率由 44% 降至 5% 以下)研究者未曾稽核,其產生
|
||||||
|
過程亦無歸檔可回溯。
|
||||||
|
|
||||||
|
**當時的判斷**:投稿未上即罷;若上,正式研究由研究者自行
|
||||||
|
設計與執行。事實上正式研究即是如此重做的。
|
||||||
|
|
||||||
|
## 三、被沿用的成果
|
||||||
|
|
||||||
|
- **歌詞捕捉檔**:先導研究蒐集的 lyrics.json(684 首,
|
||||||
|
2018–2025)以 `excludes/` 的私人腳本匯入歌詞快取,只取
|
||||||
|
識別欄位與歌詞本文,先導的分析欄位一概不匯入;出處記於
|
||||||
|
`data/captures/lyrics-provenance.csv`,method 欄標
|
||||||
|
`pilot-import`(見 `decision-log.md` 2026-07-31、08-02 條)。
|
||||||
|
- **假說方向**:女性力量語彙的挪用與污染,成為正式研究的
|
||||||
|
研究問題。
|
||||||
|
- **粒度選擇**:正式研究鎖定 thematic keywords 這一粒度,
|
||||||
|
係繼承先導研究三種粒度的比較結果(keywords 過碎、themes
|
||||||
|
過早抽象),且於執行前鎖定以防事後擇優。
|
||||||
|
- **`women-power` 一詞的來歷**:考據先導研究的 local agent
|
||||||
|
存檔可知,其第一步指令含數十個範例 thematic keywords,
|
||||||
|
其中即有 women-power——為當時協作的 Claude Code 依研究者
|
||||||
|
長期表達的關注主動加入(研究者端播種,非明示指定);先導
|
||||||
|
的標籤 `women-power-and-empowerment` 則是第三步強制合併
|
||||||
|
兩個關鍵字的管線人工產物。正式研究的探針因此指向被播種的
|
||||||
|
本詞 `women-power`,不用合併假影(見 `decision-log.md`
|
||||||
|
2026-08-04 條)。**附帶認清:先導第一步並非零語意提示,
|
||||||
|
此即正式研究「提示詞只定格式、不定語意」設計所矯正者。**
|
||||||
|
- **簿記容量的教訓**:先導研究九百餘詞可以在單一回應內完成
|
||||||
|
分組,正式研究的 5,999 個關鍵字則四種模型六次執行全部未
|
||||||
|
通過完整分割驗證,遂改用詞向量嵌入+確定性分群(見
|
||||||
|
`decision-log.md` 2026-08-06 條)。
|
||||||
|
|
||||||
|
## 四、被棄用的成果
|
||||||
|
|
||||||
|
以下先導研究的產物未進入正式研究,論文亦未引用:
|
||||||
|
|
||||||
|
- **genuine/peripheral/fake 三分類與「44% 假女性力量」**:
|
||||||
|
構念與分母皆與正式研究不同(正式研究為 145 首女性力量群
|
||||||
|
歌曲中 111 首有性別問題),兩者不可對讀。
|
||||||
|
- **五種「假女性力量」類型(A–E)**:正式研究改由儀器分三個
|
||||||
|
發話脈絡各自歸納,得 43 條問題樣態(`results/patterns.csv`)。
|
||||||
|
- **pussy 一詞的修辭分類**:正式研究範圍收斂至語彙的挪用與
|
||||||
|
污染,未納入。
|
||||||
|
- **frame-aware(框架感知)提示修正法**:先導研究以框架判準
|
||||||
|
寫進提示以提高準確率;正式研究刻意不定義編碼,因為研究
|
||||||
|
對象正是 LLM 未受引導的自然編碼——把框架寫進提示,即無法
|
||||||
|
再以其輸出為批判對象。兩者的設計方向相反,故未沿用。
|
||||||
|
- **三個獨立 LLM subagent+人工仲裁的流程**:正式研究改以
|
||||||
|
Anthropic API 逐首獨立呼叫,原因是實測證實 subagent 會繼承
|
||||||
|
CLAUDE.md 與環境資訊,context 無法僅憑定義檔重現(見
|
||||||
|
`decision-log.md` 2026-07-30 條)。
|
||||||
|
|
||||||
|
## 五、它如何促成正式研究的設計
|
||||||
|
|
||||||
|
先導研究的根本限制不在結論對錯,而在**不可稽核**:沒有定義檔
|
||||||
|
快照、沒有原始輸出歸檔、沒有參數紀錄,因此任何一個數字都無法
|
||||||
|
回溯到產生它的那一次執行。正式研究的幾項設計正是針對這一點:
|
||||||
|
|
||||||
|
- 所有 LLM 步驟改以 API script 執行,每次執行自我完備歸檔於
|
||||||
|
`runs/`(定義檔快照、原始輸出、model ID 與參數、批次 ID、
|
||||||
|
token 用量);
|
||||||
|
- 可逐項機械比對的判斷採三次執行+多數決,並量測其穩定性
|
||||||
|
(見 `reliability.md`);
|
||||||
|
- 定義檔只規定任務形狀,不給主題定義、判準或範例;
|
||||||
|
- 每一筆編碼必附歌詞引述,供逐筆查核;
|
||||||
|
- 輸出的契約遵從另行查核(見 `output-validation.md`)。
|
||||||
|
|
||||||
|
換言之,正式研究對 LLM 輸出所採取的「不信任、須查核」立場,
|
||||||
|
其第一個案例就是先導研究本身。
|
||||||
|
|
||||||
|
## 六、現存的紀錄
|
||||||
|
|
||||||
|
- 投稿時的先導研究摘要:已於 git 歷史中(`paper/abstract.md`
|
||||||
|
的前身),樹中不另存副本——摘要只是產物,不足以記述過程,
|
||||||
|
故另立本檔。
|
||||||
|
- 先導研究的歌詞捕捉檔沿用事實:`data/captures/lyrics-provenance.csv`。
|
||||||
|
- 相關決策條目:`decision-log.md` 2026-07-30(不與先導比較、
|
||||||
|
不用 subagent)、07-31 與 08-02(歌詞沿用與私人匯入腳本)、
|
||||||
|
08-04(women-power 來歷考據)、08-06(詞彙表改用詞向量分群)。
|
||||||
+48
-23
@@ -1,13 +1,11 @@
|
|||||||
# 專案目錄結構
|
# 專案目錄結構
|
||||||
|
|
||||||
(2026-07-30 討論定案;2026-07-31 更新為 tools/ 子專案與
|
(2026-07-30 討論定案;2026-07-31 更新為 tools/ 子專案與
|
||||||
SQLite 工作儲存架構)
|
SQLite 工作儲存架構;2026-08-17 依完成後的現況更新)
|
||||||
|
|
||||||
```
|
```
|
||||||
pop-fem-audit/
|
pop-fem-audit/
|
||||||
├── README.md # 專案說明、重現步驟
|
├── README.md # 專案說明
|
||||||
├── CLAUDE.md # 極簡工作規範(subagent 會讀到,
|
|
||||||
│ # 絕不放理論、codebook、預期結果)
|
|
||||||
├── .gitignore # captures/lyrics/、.env、scratch
|
├── .gitignore # captures/lyrics/、.env、scratch
|
||||||
├── data/ # 依生命週期分層(文字格式)
|
├── data/ # 依生命週期分層(文字格式)
|
||||||
│ ├── source/ # 源頭:手放後不動
|
│ ├── source/ # 源頭:手放後不動
|
||||||
@@ -19,17 +17,24 @@ pop-fem-audit/
|
|||||||
│ │ └── lyrics/ # 歌詞 .txt 快取
|
│ │ └── lyrics/ # 歌詞 .txt 快取
|
||||||
│ │ # (gitignored,版權)
|
│ │ # (gitignored,版權)
|
||||||
│ ├── manual/ # 人工著作:只由研究者手寫
|
│ ├── manual/ # 人工著作:只由研究者手寫
|
||||||
│ │ # (黃金標準編碼等)
|
│ │ ├── coding-corrections.csv # 編碼與引述的校對表
|
||||||
|
│ │ └── performer-gender-corrections.csv # 演唱聲音性別的
|
||||||
|
│ │ # 手工修正
|
||||||
│ └── derived/ # 衍生:只由 build-db 寫入
|
│ └── derived/ # 衍生:只由 build-db 寫入
|
||||||
│ ├── songs.csv # 歌曲報表(人讀;進 git)
|
│ ├── songs.csv # 歌曲報表(人讀;進 git)
|
||||||
│ └── artists.csv # 歌手報表(人讀;進 git)
|
│ └── artists.csv # 歌手報表(人讀;進 git)
|
||||||
├── prompts/ # LLM 定義檔(逐字作為 system prompt)
|
├── prompts/ # LLM 定義檔(逐字作為 system prompt)
|
||||||
│ └── <步>-<次步>-<task>.md # 01-tag.md、03-code.md
|
│ └── <步><次步>-<task>.md # 1-tag.md、3a-code.md、
|
||||||
│ # (步內僅一個執行時省略次步)
|
│ # 4-group.md、5a-read.md、
|
||||||
|
│ # 5b-consolidate.md、
|
||||||
|
│ # 5c-synthesize.md、
|
||||||
|
│ # 5d-annotate.md
|
||||||
|
│ # (次步以字母標示,與論文正文
|
||||||
|
│ # 的步驟編號一致;步內僅一個
|
||||||
|
│ # 執行時省略次步)
|
||||||
│ # 不帶版本號,版本即 git 歷史
|
│ # 不帶版本號,版本即 git 歷史
|
||||||
│ # (編號的所指是工序:確定性
|
│ # (編號的所指是工序:確定性
|
||||||
│ # 的步驟 2 無定義檔仍佔一號;
|
│ # 的步驟 2 無定義檔仍佔一號)
|
||||||
│ # 補零只為排序)
|
|
||||||
├── tools/ # 輔助工具子專案(src-layout)
|
├── tools/ # 輔助工具子專案(src-layout)
|
||||||
│ ├── pyproject.toml # 發行名 pop-fem-audit-tools;
|
│ ├── pyproject.toml # 發行名 pop-fem-audit-tools;
|
||||||
│ │ # pip install -e tools/ 安裝
|
│ │ # pip install -e tools/ 安裝
|
||||||
@@ -52,6 +57,9 @@ pop-fem-audit/
|
|||||||
│ │ │ ├── cluster_keywords.py # pool the tagging runs'
|
│ │ │ ├── cluster_keywords.py # pool the tagging runs'
|
||||||
│ │ │ │ # keywords and cluster them
|
│ │ │ │ # keywords and cluster them
|
||||||
│ │ │ │ # into the codes (step 2)
|
│ │ │ │ # into the codes (step 2)
|
||||||
|
│ │ │ ├── tally_codings.py # settle step 3 by majority
|
||||||
|
│ │ │ ├── tally_groups.py # settle step 4 by majority
|
||||||
|
│ │ │ ├── tally_annotations.py # settle step 5d by majority
|
||||||
│ │ │ └── run_llm.py # API 執行器:一份定義檔+一份輸入
|
│ │ │ └── run_llm.py # API 執行器:一份定義檔+一份輸入
|
||||||
│ │ │ # →歸檔至指定目錄(Batch API);
|
│ │ │ # →歸檔至指定目錄(Batch API);
|
||||||
│ │ │ # 多次執行的計票由獨立子命令承擔
|
│ │ │ # 多次執行的計票由獨立子命令承擔
|
||||||
@@ -62,22 +70,34 @@ pop-fem-audit/
|
|||||||
│ └── tests/ # 單元測試(unittest)
|
│ └── tests/ # 單元測試(unittest)
|
||||||
├── runs/ # 現行執行的完整稽核紀錄(進 git;
|
├── runs/ # 現行執行的完整稽核紀錄(進 git;
|
||||||
│ │ # 重跑同一 run 須明示 --replace)
|
│ │ # 重跑同一 run 須明示 --replace)
|
||||||
│ ├── <步驟名>/ # 一步一個目錄(如 03-code)
|
│ ├── <步驟名>/ # 一步一個目錄(1-tag、3a-code、
|
||||||
|
│ │ # 4-group、5a-read、
|
||||||
|
│ │ # 5b-consolidate、
|
||||||
|
│ │ # 5c-synthesize、5d-annotate)
|
||||||
│ │ └── run<N>/ # LLM 步驟:每個 run 一份自我
|
│ │ └── run<N>/ # LLM 步驟:每個 run 一份自我
|
||||||
│ │ ├── prompt.md # 完備歸檔(定義檔快照)
|
│ │ ├── prompt.md # 完備歸檔(定義檔快照)
|
||||||
│ │ ├── output.jsonl # 該次執行原始輸出
|
│ │ ├── output.jsonl # 該次執行原始輸出
|
||||||
│ │ └── meta.json # model ID、temperature、時間戳、
|
│ │ └── meta.json # model ID、temperature、時間戳、
|
||||||
│ │ # batch ID、token 用量
|
│ │ # batch ID、token 用量
|
||||||
│ └── 02-cluster/ # 確定性步驟:無執行變異,
|
│ └── 2-cluster/ # 確定性步驟:無執行變異,
|
||||||
│ # 不分 run<N> 層
|
│ # 不分 run<N> 層
|
||||||
├── results/ # 論文引用的報表 CSV(export 產出;
|
├── results/ # 論文引用的定案表 CSV(計票子命令
|
||||||
│ # 「可再生仍 commit」的唯一例外)
|
│ │ # 產出;「可再生仍 commit」的例外)
|
||||||
|
│ ├── codings.csv # 步驟 3 定案編碼
|
||||||
|
│ ├── groups.csv # 步驟 4 定案編碼群
|
||||||
|
│ ├── patterns.csv # 步驟 5c 定案樣態表
|
||||||
|
│ ├── annotations.csv # 步驟 5d 定案歌×樣態
|
||||||
|
│ └── pattern-matrix.csv # 前四者的人讀寬表
|
||||||
├── docs/
|
├── docs/
|
||||||
|
│ ├── conventions.md # 常設工作規範(原 CLAUDE.md;
|
||||||
|
│ │ # 移入 docs/ 使 subagent 不繼承)
|
||||||
│ ├── research-plan.md # 研究步驟規劃(本檔之姊妹篇)
|
│ ├── research-plan.md # 研究步驟規劃(本檔之姊妹篇)
|
||||||
│ ├── project-structure.md # 本檔
|
│ ├── project-structure.md # 本檔
|
||||||
│ ├── codebook.md # 人工編碼手冊(版本由 git 管理)
|
│ ├── output-validation.md # LLM 輸出的契約查核紀錄
|
||||||
│ ├── decision-log.md # 決策日誌:每次改定義檔的原因
|
│ ├── decision-log.md # 決策日誌:每次改定義檔的原因
|
||||||
│ ├── run-costs.md # 每次執行的 token 用量與費用
|
│ ├── run-costs.md # 每次執行的 token 用量與費用
|
||||||
|
│ ├── reliability.md # 信度:量測方式與結果
|
||||||
|
│ ├── pilot-study.md # 先導研究的來歷與地位
|
||||||
│ └── methodology.md # 方法細節(全文方法節底稿;
|
│ └── methodology.md # 方法細節(全文方法節底稿;
|
||||||
│ # 映射分析方法須在看結果前寫定)
|
│ # 映射分析方法須在看結果前寫定)
|
||||||
└── paper/
|
└── paper/
|
||||||
@@ -94,12 +114,15 @@ pop-fem-audit/
|
|||||||
- **`prompts/` 檔名不帶版本號**:版本即 git 歷史,失敗的
|
- **`prompts/` 檔名不帶版本號**:版本即 git 歷史,失敗的
|
||||||
版本不保留;論文引用的單位是 `runs/` 內隨執行保存的定義檔
|
版本不保留;論文引用的單位是 `runs/` 內隨執行保存的定義檔
|
||||||
快照(每個執行目錄自我完備),不需檔名可指的版本名。
|
快照(每個執行目錄自我完備),不需檔名可指的版本名。
|
||||||
- **工作儲存的資料表**:`songs`、`chart_entries`、`artists`、
|
- **工作儲存的資料表**:`songs`(含 `performer_gender`=演唱
|
||||||
`song_artists`、`codings`(定案編碼:一歌一標籤一列,`quotes`
|
聲音的性別)、`chart_entries`、`artists`、`song_artists`、
|
||||||
存該標籤所據的歌詞引述,多句以 `|` 相接)。定案表
|
`codings`(定案編碼:一歌一標籤一列,`quotes` 存該標籤所據的
|
||||||
`results/codings.csv` 經 `build-db --codings` 匯入,與其餘資料
|
歌詞引述,多句以 `|` 相接)、`groups`(語意編碼群)、
|
||||||
同一交易,儲存不會半建;詳見 `research-plan.md`「資料儲存與
|
`patterns`(深讀樣態)、`annotations`(歌×樣態定案矩陣)。
|
||||||
模型」。
|
各定案表經 `build-db` 的 `--codings`、`--groups`、
|
||||||
|
`--patterns`、`--annotations` 匯入,性別修正經
|
||||||
|
`--gender-corrections` 套用,與其餘資料同一交易,儲存不會
|
||||||
|
半建;詳見 `research-plan.md`「資料儲存與模型」。
|
||||||
- **Commit 判準**:能由「committed 輸入+程式」決定性再生者不
|
- **Commit 判準**:能由「committed 輸入+程式」決定性再生者不
|
||||||
commit(SQLite 工作儲存、LLM 輸入檔);源頭、捕捉、人工著作
|
commit(SQLite 工作儲存、LLM 輸入檔);源頭、捕捉、人工著作
|
||||||
一律以文字 commit。「可再生仍 commit」的例外有二:
|
一律以文字 commit。「可再生仍 commit」的例外有二:
|
||||||
@@ -109,6 +132,8 @@ pop-fem-audit/
|
|||||||
- **設定**經 pydantic-settings 統一:`.env`(gitignored,範本
|
- **設定**經 pydantic-settings 統一:`.env`(gitignored,範本
|
||||||
`tools/.env.example`)供應 `SQLALCHEMY_DATABASE_URL` 與
|
`tools/.env.example`)供應 `SQLALCHEMY_DATABASE_URL` 與
|
||||||
`ANTHROPIC_API_KEY`,絕不寫入 repo。
|
`ANTHROPIC_API_KEY`,絕不寫入 repo。
|
||||||
- **CLAUDE.md 極簡**:實測證實 Claude Code subagent 會繼承專案
|
- **不設 CLAUDE.md**:實測證實 Claude Code subagent 會繼承專案
|
||||||
CLAUDE.md 全文,故其中只放工作流程規則,領域知識一律放
|
CLAUDE.md 全文(原本因此只放極簡工作規則),2026-08-18 進一步
|
||||||
`docs/`(subagent 不會自動讀到)。
|
將其移為 `docs/conventions.md`——docs/ 不會自動注入 subagent
|
||||||
|
的 context,盲判型 agent 便不會看到工作規範;主會話動手前
|
||||||
|
自行閱讀之。
|
||||||
|
|||||||
@@ -0,0 +1,161 @@
|
|||||||
|
# 信度:量測方式與結果
|
||||||
|
|
||||||
|
(2026-08-16 量測。對象為步驟 3a 的三份執行歸檔
|
||||||
|
`runs/3a-code/run1`–`run3`,與步驟 5d 的三份執行歸檔
|
||||||
|
`runs/5d-annotate/run1`–`run3`;數字由原始輸出直接計算。)
|
||||||
|
|
||||||
|
## 一、本研究的信度是什麼
|
||||||
|
|
||||||
|
**信度(reliability)** 問的是:同一個量測重複做,會不會得到
|
||||||
|
同樣的結果。它管的是**一致性**,不管對錯。一把每次都少兩公斤
|
||||||
|
的秤,信度很好、效度很差。**效度(validity)** 問的則是:量到
|
||||||
|
的是不是想量的東西。
|
||||||
|
|
||||||
|
Krippendorff 依產生資料的設計,把信度分成三型
|
||||||
|
(Krippendorff, 2004;Hayes & Krippendorff, 2007):
|
||||||
|
|
||||||
|
- **穩定性(stability)**:同一位觀察者、同樣的材料,重複
|
||||||
|
量測是否一致(test–retest,即 intra-observer)。
|
||||||
|
- **可複製性(reproducibility)**:不同觀察者各自獨立量測
|
||||||
|
是否一致,即一般所稱的**編碼者間信度**
|
||||||
|
(inter-coder reliability)。Krippendorff 認為這是內容
|
||||||
|
分析中最強、也最可行的一型,因為它排除了「這些類目只是
|
||||||
|
某一個人特異的讀法」的疑慮。
|
||||||
|
- **準確性(accuracy)**:與已知標準比對。
|
||||||
|
|
||||||
|
**本研究三次獨立執行量到的是第一型。**同一個模型、同一份
|
||||||
|
定義檔、同一批輸入,重複三次,量到的是這個編碼者自己前後
|
||||||
|
一不一致。第二型在本設計中**結構上不存在**:編碼者只有一位,
|
||||||
|
不同的模型版本(Sonnet 4.6、Fable 5)也不是「不同的編碼者」,
|
||||||
|
而是不同的儀器。第三型**未量測**:沒有黃金標準可比。
|
||||||
|
|
||||||
|
以工程的語言說更精確:這裏量的是**儀器的重複性
|
||||||
|
(repeatability)**,亦即**隨機性有多大**;計量學上稱之為
|
||||||
|
**精密度(precision)**,而**準確度(accuracy)** 未經量測。
|
||||||
|
若儀器是確定性的,重複性無須量測;但 LLM 不是——同樣的提示、
|
||||||
|
同樣的輸入,三次會給出不同結果,因此重複性是必須報告的儀器
|
||||||
|
規格,而非慣例儀式。
|
||||||
|
|
||||||
|
這一點與 LLM 標註的近期文獻一致:同一模型重複取樣量測的是
|
||||||
|
自我一致性(self-consistency),而重複執行後取多數決可提升
|
||||||
|
標註穩定度(如 Prompt Stability Scoring,arXiv:2407.02039)。
|
||||||
|
|
||||||
|
## 二、三個常用指標
|
||||||
|
|
||||||
|
以「這首歌有沒有這個碼」這種是非題為例:
|
||||||
|
|
||||||
|
- **百分比一致率**:兩次判斷相同的格子佔全部格子的比例。
|
||||||
|
最直觀,但會被「碰巧同意」灌水——當九成五的格子都是
|
||||||
|
「無」,兩次隨便判也會有九成以上一致。
|
||||||
|
- **Cohen's κ(kappa)**:把「碰巧同意」的部分扣掉之後的
|
||||||
|
一致度,兩位編碼者用。0 表示與隨機無異,1 表示完全一致。
|
||||||
|
慣例上 0.61–0.80 稱為 substantial、0.81 以上稱為
|
||||||
|
almost perfect(Landis & Koch 1977 的分級)。
|
||||||
|
- **Krippendorff's α**:可處理兩位以上編碼者、缺漏值與
|
||||||
|
不同尺度,是內容分析文獻最常被要求的指標。慣例門檻為
|
||||||
|
α ≥ 0.800 可作結論、0.667–0.800 只能作暫定結論
|
||||||
|
(Krippendorff 的建議)。
|
||||||
|
|
||||||
|
三者裏,**百分比一致率會高估、κ 與 α 才是可比較的數字**。
|
||||||
|
|
||||||
|
補充一個常被誤解的方向:κ 的著名「悖論」
|
||||||
|
(Feinstein & Cicchetti, 1990)是**類別分佈極不平衡時,
|
||||||
|
高一致率反而算出很低的 κ**——偏斜會壓低 κ,不會抬高它。
|
||||||
|
本研究的正例只佔 16.5%(89,183 格中約 14,700 格有碼),
|
||||||
|
在這種偏斜下 κ 仍達 0.93,屬於保守估計,不是被大量「雙方
|
||||||
|
都判無」灌出來的。
|
||||||
|
|
||||||
|
## 三、實測結果
|
||||||
|
|
||||||
|
### 步驟 3a(編碼:883 首 × 101 個碼 = 89,183 格,執行三次)
|
||||||
|
|
||||||
|
| 兩次執行 | 百分比一致率 | Cohen's κ | Jaccard |
|
||||||
|
|---|---:|---:|---:|
|
||||||
|
| run1 × run2 | 98.10% | 0.931 | 0.891 |
|
||||||
|
| run1 × run3 | 98.09% | 0.931 | 0.891 |
|
||||||
|
| run2 × run3 | 98.26% | 0.937 | 0.900 |
|
||||||
|
|
||||||
|
**Krippendorff's α = 0.933**(三次執行合計)。
|
||||||
|
|
||||||
|
三票計票的分佈:三票全同 13,500 格、兩票 1,163 格、
|
||||||
|
僅一票 1,310 格(未達門檻而剔除)。**定案的 14,663 個編碼
|
||||||
|
中,92.1% 是三次全數同意的**。
|
||||||
|
|
||||||
|
(此處由原始輸出計算,未套用 936 筆人工校對表;論文引用的
|
||||||
|
定案數 14,664 為校對後的結果,差 1 筆。)
|
||||||
|
|
||||||
|
### 步驟 5d(樣態標註:111 首 × 各自適用樣態 = 3,708 格,執行三次)
|
||||||
|
|
||||||
|
| 兩次執行 | 百分比一致率 | Cohen's κ | Jaccard |
|
||||||
|
|---|---:|---:|---:|
|
||||||
|
| run1 × run2 | 97.87% | 0.933 | 0.898 |
|
||||||
|
| run1 × run3 | 98.17% | 0.942 | 0.911 |
|
||||||
|
| run2 × run3 | 97.87% | 0.933 | 0.898 |
|
||||||
|
|
||||||
|
**Krippendorff's α = 0.936**。三票全同 675 格、兩票 62 格、
|
||||||
|
僅一票 51 格;**定案的 737 筆歸屬中,91.6% 三次全同**。
|
||||||
|
|
||||||
|
### 怎麼讀這些數字
|
||||||
|
|
||||||
|
κ 與 α 都在 0.93 上下,以慣例分級屬於 almost perfect,
|
||||||
|
也高於 α ≥ 0.800 的門檻。用白話說:**這個編碼者重複三次,
|
||||||
|
判斷幾乎不變;會左右搖擺的,是那一成左右的邊緣案例,而
|
||||||
|
三票多數決正是為它們設計的**。
|
||||||
|
|
||||||
|
Jaccard(只看「有標到」的格子,忽略雙方都沒標的)約 0.89–0.91,
|
||||||
|
比百分比一致率低而更誠實——因為 82% 的格子是雙方都判「無」,
|
||||||
|
那些一致並不費力。
|
||||||
|
|
||||||
|
## 四、隨機性的規模與三票制的作用
|
||||||
|
|
||||||
|
信度數字回答的實際問題是:**這台儀器的隨機性,大到會不會
|
||||||
|
改變結論?**
|
||||||
|
|
||||||
|
步驟 3a 的 89,183 格中,搖擺的格子共 2,473 格(兩票 1,163、
|
||||||
|
一票 1,310),佔 2.8%。若只執行一次即定案,這批格子當中約
|
||||||
|
一半會成為誤收、另一半會成為漏收。三票多數決把兩票以上者
|
||||||
|
收入、一票者剔除,處理的正是這一批。
|
||||||
|
|
||||||
|
三票制並非消除隨機性,而是**把隨機性往案例原本的傾向推**。
|
||||||
|
設某個邊緣案例的符合程度為 p,單次執行以機率 p 標出,三次
|
||||||
|
多數決則以 p³+3p²(1−p) 標出:p=0.9 者由 0.9 提高到 0.972,
|
||||||
|
p=0.1 者由 0.1 壓低到 0.028,而 p=0.5 者仍是 0.5——真正
|
||||||
|
模稜兩可的案例,任何票制都救不了。
|
||||||
|
|
||||||
|
以此規模判斷,結論層的三條帶狀結構(女性力量與陽剛群共現、
|
||||||
|
與脆弱群互斥、與厭女群獨立)不可能由這個量級的雜訊翻轉。
|
||||||
|
反過來說,若一致率只有 0.6,同一組結論就不能採信——信度
|
||||||
|
數字的用途在此,而不在滿足慣例。
|
||||||
|
|
||||||
|
## 五、這些數字證明了什麼、不能證明什麼
|
||||||
|
|
||||||
|
**能證明**:量測是穩定的,結果不是單次抽樣的偶然;
|
||||||
|
三票多數決確實只在少數邊緣格子上發揮作用(約 3%)。
|
||||||
|
|
||||||
|
**不能證明**:判斷是正確的。三次共用同一個模型與同一份
|
||||||
|
先驗,能濾掉隨機噪音,濾不掉系統性偏誤。本研究已發現三個
|
||||||
|
三次一致的錯誤可作實例:Sonnet 4.6 穩定地將 Women Power
|
||||||
|
讀為 Women+Power;深讀階段將〈Cowgirls〉中握韁繩者判反、
|
||||||
|
將〈Happen To Me〉分屬兩段的引文接成同一動機。一把每次都
|
||||||
|
量出同樣偏差的尺,重複性滿分,準確度為零。
|
||||||
|
|
||||||
|
**結構上沒有**:編碼者間信度。要有它,需要第二個獨立的
|
||||||
|
觀察者——另一個模型,或研究者人工編一小批對照。本研究
|
||||||
|
未做,列為限制。
|
||||||
|
|
||||||
|
## 六、計算方式與依據
|
||||||
|
|
||||||
|
- **分析單位**:每一個「(歌曲, 編碼)」格為一個單位,是非題
|
||||||
|
(有標/無標)。步驟 3a 為 883 首 × 101 碼 = 89,183 格;
|
||||||
|
步驟 5d 為各歌適用樣態數合計 3,708 格。
|
||||||
|
- **Cohen's κ**:兩次執行的 2×2 表,κ=(p_o−p_e)/(1−p_e),
|
||||||
|
p_e 由兩次各自的邊際比例相乘求得。
|
||||||
|
- **Krippendorff's α(名目、三位「編碼者」、無缺漏)**:
|
||||||
|
α=1−D_o/D_e;某單位若有 v 次標為「有」,其不一致配對數為
|
||||||
|
v(3−v),D_o 為其總和除以總配對數,D_e 由整體「有/無」的
|
||||||
|
邊際比例求得。
|
||||||
|
- **門檻依據**:κ 的分級為 Landis & Koch (1977)
|
||||||
|
(0.61–0.80 substantial、0.81–1.00 almost perfect);
|
||||||
|
α 的門檻為 Krippendorff 的建議(≥0.800 可作結論、
|
||||||
|
0.667–0.800 僅能作暫定結論)。
|
||||||
|
- 數字由原始執行輸出直接計算,未套用 936 筆人工校對表。
|
||||||
+17
-12
@@ -15,10 +15,10 @@
|
|||||||
script 執行。LLM 步驟以 Python script 呼叫 Anthropic
|
script 執行。LLM 步驟以 Python script 呼叫 Anthropic
|
||||||
Messages API(個人 Console 帳號、Batch API 五折),定義
|
Messages API(個人 Console 帳號、Batch API 五折),定義
|
||||||
檔逐字作為 system prompt。可逐項機械比對的判斷(步驟 3
|
檔逐字作為 system prompt。可逐項機械比對的判斷(步驟 3
|
||||||
編碼、步驟 4 選群、步驟 5-4 樣態標註)採「同一定義檔
|
編碼、步驟 4 選群、步驟 5d 樣態標註)採「同一定義檔
|
||||||
獨立執行三次+多數決」,計票由確定性子命令完成。自由
|
獨立執行三次+多數決」,計票由確定性子命令完成。自由
|
||||||
生成(步驟 1 自由標註)兩次執行全數進池。步驟 5-1 至
|
生成(步驟 1 自由標註)兩次執行全數進池。步驟 5a 至
|
||||||
5-3 為質性閱讀協定:三次獨立閱讀為分析者三角檢核、
|
5c 為質性閱讀協定:三次獨立閱讀為分析者三角檢核、
|
||||||
逐首整合、樣態統整產出草稿,不適用投票與仲裁。詞彙表
|
逐首整合、樣態統整產出草稿,不適用投票與仲裁。詞彙表
|
||||||
不經 LLM,由詞向量嵌入+確定性分群產生。驗證結果不符
|
不經 LLM,由詞向量嵌入+確定性分群產生。驗證結果不符
|
||||||
預期則修訂定義檔重跑該循環,絕不手改結果。
|
預期則修訂定義檔重跑該循環,絕不手改結果。
|
||||||
@@ -60,8 +60,8 @@
|
|||||||
(`data/derived/`,進 git),與 SQLite 同交易語意。
|
(`data/derived/`,進 git),與 SQLite 同交易語意。
|
||||||
- 報表:論文引用的定案表進 `results/`——
|
- 報表:論文引用的定案表進 `results/`——
|
||||||
`codings.csv`(步驟 3 定案編碼)、`groups.csv`
|
`codings.csv`(步驟 3 定案編碼)、`groups.csv`
|
||||||
(步驟 4 定案編碼群)、`patterns.csv`(步驟 5-3 定案
|
(步驟 4 定案編碼群)、`patterns.csv`(步驟 5c 定案
|
||||||
樣態表)、`annotations.csv`(步驟 5-4 定案歌×樣態
|
樣態表)、`annotations.csv`(步驟 5d 定案歌×樣態
|
||||||
矩陣)。衍生與報表為「可再生仍 commit」的例外,理由:
|
矩陣)。衍生與報表為「可再生仍 commit」的例外,理由:
|
||||||
引用穩定性、審稿人零門檻、撰稿期數字變動可 diff。
|
引用穩定性、審稿人零門檻、撰稿期數字變動可 diff。
|
||||||
- **資料模型**:`songs`(含 lyrics、`performer_gender`——
|
- **資料模型**:`songs`(含 lyrics、`performer_gender`——
|
||||||
@@ -76,8 +76,8 @@
|
|||||||
**絕不進 LLM 輸入**——LLM 任務的 user message 維持
|
**絕不進 LLM 輸入**——LLM 任務的 user message 維持
|
||||||
歌詞-only 或管線中間產物-only,避免光環偏誤。
|
歌詞-only 或管線中間產物-only,避免光環偏誤。
|
||||||
- **Pilot 歌詞沿用(私人匯入,不進發布管線)**:先導研究
|
- **Pilot 歌詞沿用(私人匯入,不進發布管線)**:先導研究
|
||||||
捕捉檔(lyrics.json,684 首,2018–2025)以私人腳本
|
捕捉檔(lyrics.json,684 首,2018–2025)以 `excludes/`
|
||||||
匯入歌詞快取;讀者的重現路徑純粹是
|
的私人腳本匯入歌詞快取;讀者的重現路徑純粹是
|
||||||
`fetch-lyrics`;沿用之事實記於
|
`fetch-lyrics`;沿用之事實記於
|
||||||
`data/captures/lyrics-provenance.csv`(進 git)。
|
`data/captures/lyrics-provenance.csv`(進 git)。
|
||||||
- **子命令**(`pop-fem-audit-tools <cmd>`):`build-db`
|
- **子命令**(`pop-fem-audit-tools <cmd>`):`build-db`
|
||||||
@@ -105,15 +105,16 @@
|
|||||||
定案 `results/groups.csv`;wp/fe 與各編碼、各編碼群之
|
定案 `results/groups.csv`;wp/fe 與各編碼、各編碼群之
|
||||||
關聯統計(BH-FDR 校正)入論文。
|
關聯統計(BH-FDR 校正)入論文。
|
||||||
5. **女性主義問題之質性深讀(步驟 5)**:對 wp∪fe 145 首
|
5. **女性主義問題之質性深讀(步驟 5)**:對 wp∪fe 145 首
|
||||||
——5-1 逐首盲讀(僅歌詞全文)×3;5-2 逐首整合(收斂
|
——5a 逐首盲讀(僅歌詞全文)×3;5b 逐首整合(收斂
|
||||||
註記、主清單限兩讀以上);5-3 樣態統整(全體基底+
|
註記、主清單限兩讀以上);5c 樣態統整(全體基底+
|
||||||
男聲/女聲/混合三個發話脈絡分組);5-4 樣態標註——
|
男聲/女聲/混合三個發話脈絡分組);5d 樣態標註——
|
||||||
以分組樣態表逐首標註「有問題」的 111 首,×3+多數決
|
以分組樣態表逐首標註「有問題」的 111 首,×3+多數決
|
||||||
(`tally-annotations`),定案 `results/patterns.csv` 與
|
(`tally-annotations`),定案 `results/patterns.csv` 與
|
||||||
`results/annotations.csv`。
|
`results/annotations.csv`。
|
||||||
|
|
||||||
定義檔命名 `prompts/<步>-<次步>-<task>.md`,編號的所指是
|
定義檔命名 `prompts/<步><次步>-<task>.md`,次步以字母標示
|
||||||
工序而非定義檔(確定性的第 2 步無定義檔仍佔編號);檔名
|
(與論文正文的步驟編號一致);編號的所指是工序而非定義檔
|
||||||
|
(確定性的第 2 步無定義檔仍佔編號);檔名
|
||||||
不帶版本號——版本即 git 歷史;每次執行的定義檔快照隨
|
不帶版本號——版本即 git 歷史;每次執行的定義檔快照隨
|
||||||
`runs/` 自我完備,token 費用逐筆記於 `docs/run-costs.md`。
|
`runs/` 自我完備,token 費用逐筆記於 `docs/run-costs.md`。
|
||||||
|
|
||||||
@@ -129,3 +130,7 @@
|
|||||||
- 歌詞受版權保護:完整歌詞不進 git
|
- 歌詞受版權保護:完整歌詞不進 git
|
||||||
(`data/captures/lyrics/` gitignored),論文與 repo 只留
|
(`data/captures/lyrics/` gitignored),論文與 repo 只留
|
||||||
分析所引摘錄。
|
分析所引摘錄。
|
||||||
|
- 工程支援(寫 script、style check、審稿)用公司訂閱的
|
||||||
|
Claude Code;進論文的分析 token 由個人 API 帳號支付。
|
||||||
|
- Positionality statement 寫入論文(53 歲、長年婦運者、
|
||||||
|
資深工程師、離開學術圈多年、因整理榜單而起)。
|
||||||
|
|||||||
+56
-48
@@ -10,53 +10,61 @@ $3/$15、opus-4-6 $5/$25、opus-5 與 fable-5 $10/$50)
|
|||||||
被 `--replace` 取代的執行以「已取代」標記,數字保留供
|
被 `--replace` 取代的執行以「已取代」標記,數字保留供
|
||||||
總支出核算。
|
總支出核算。
|
||||||
|
|
||||||
| 日期 | 步驟 | 執行 | 模型 | 批次 ID | 耗時 | input | output | 費用 (USD) | 狀態 |
|
「步驟」欄的名稱於 2026-08-18 一律換為現行編號(見
|
||||||
|---|---|---|---|---|---|---:|---:|---:|---|
|
`decision-log.md` 當日條目),執行當時的名稱記於末欄
|
||||||
| 2026-08-04 | 01-01-01-tag | run1 | claude-sonnet-4-6 | msgbatch_01PTACDQMr8M6ahnshbedjtB | 6 分 27 秒 | 763,318 | 338,661 | $3.68 | 已取代(浮水印清洗與防圍欄修訂後重跑) |
|
「原步驟名」,未曾改名者留空。已刪除的工序——步驟 2 的
|
||||||
| 2026-08-05 | 01-01-01-tag | run1 | claude-sonnet-4-6 | msgbatch_01TikJNd2pZVxzQ8SaybVthu | 4 分 57 秒 | 774,604 | 323,651 | $3.59 | 已取代(合法 JSON 修訂後重跑) |
|
LLM 合併嘗試(`01-02-01-merge`,詞彙表改採詞向量分群時
|
||||||
| 2026-08-05 | 01-01-01-tag | run1 | claude-sonnet-4-6 | msgbatch_01JFBCNqnu1cwXyqmEKLHQYF | 4 分 30 秒 | 790,370 | 328,227 | $3.65 | 已取代(song-288 平台失敗,整批重跑驗證) |
|
放棄)與步驟 3 的仲裁(`03-02-arbitration`,改採三票
|
||||||
| 2026-08-05 | 01-01-01-tag | run1 | claude-sonnet-4-6 | msgbatch_01VSDneWuSbf8mShA32jiWrX | 6 分 15 秒 | 790,370 | 326,193 | $3.63 | 已取代(措辭修訂後全體重跑) |
|
多數決時刪除)——不另發新名,步驟欄標「已廢棄」,僅存
|
||||||
| 2026-08-05 | 01-01-01-tag | run1-rescue-288 | claude-sonnet-4-6 | msgbatch_019Hq6bNVXVmp4DjRZ2cgVda | 2 分 5 秒 | 1,015 | 376 | $0.01 | 已取代(措辭修訂後全體重跑,該首原生通過)|
|
本表供總支出核算。
|
||||||
| 2026-08-05 | 01-01-01-tag | run1 | claude-sonnet-4-6 | msgbatch_01VgZ77KAPGWmuu3PnQFqZ7Q | 7 分 1 秒 | 794,913 | 326,435 | $3.64 | 現行 |
|
|
||||||
| 2026-08-05 | 01-01-01-tag | run2 | claude-sonnet-4-6 | msgbatch_01TLFey3L4fimKxcebTZQGYn | 4 分 21 秒 | 794,913 | 328,324 | $3.65 | 現行 |
|
| 日期 | 步驟 | 執行 | 模型 | 批次 ID | 耗時 | input | output | 費用 (USD) | 狀態 | 原步驟名 |
|
||||||
| 2026-08-05 | 01-02-01-merge | run1 | claude-sonnet-4-6 | msgbatch_01BvYMFmH8zrUq9SNxWSNba7 | 9 分 14 秒 | 46,454 | 60,974 | $0.53 | 已取代(完整分割驗證不過:漏 366、重複分派 511、撞名 4) |
|
|---|---|---|---|---|---|---:|---:|---:|---|---|
|
||||||
| 2026-08-05 | 01-02-01-merge | run1 | claude-opus-5 | msgbatch_018TkYpdT3DsZqq2q94FQCUN | — | 0 | 0 | $0.00 | 拒收(temperature 已棄用) |
|
| 2026-08-04 | 1-tag | run1 | claude-sonnet-4-6 | msgbatch_01PTACDQMr8M6ahnshbedjtB | 6 分 27 秒 | 763,318 | 338,661 | $3.68 | 已取代(浮水印清洗與防圍欄修訂後重跑) | 01-01-01-tag |
|
||||||
| 2026-08-05 | 01-02-01-merge | run1 | claude-opus-4-8 | msgbatch_018Wuux4adz5mWSvzJjfgLxb | — | 0 | 0 | $0.00 | 拒收(temperature 已棄用) |
|
| 2026-08-05 | 1-tag | run1 | claude-sonnet-4-6 | msgbatch_01TikJNd2pZVxzQ8SaybVthu | 4 分 57 秒 | 774,604 | 323,651 | $3.59 | 已取代(合法 JSON 修訂後重跑) | 01-01-01-tag |
|
||||||
| 2026-08-05 | 01-02-01-merge | run1 | claude-opus-4-6 | msgbatch_01Lmt2j1Txyof9CctUK7CbHA | 12 分 19 秒 | 46,454 | 54,864 | $0.80 | 已取代(圍欄違規;漏 190、發明 151、重複 11) |
|
| 2026-08-05 | 1-tag | run1 | claude-sonnet-4-6 | msgbatch_01JFBCNqnu1cwXyqmEKLHQYF | 4 分 30 秒 | 790,370 | 328,227 | $3.65 | 已取代(song-288 平台失敗,整批重跑驗證) | 01-01-01-tag |
|
||||||
| 2026-08-05 | 01-02-01-merge | run1 | claude-opus-5 | msgbatch_012SEfebTnjU5uWN6hhfnhhh | 8 分 45 秒 | 63,368 | 64,000 | $1.92 | 已取代(64k 截斷) |
|
| 2026-08-05 | 1-tag | run1 | claude-sonnet-4-6 | msgbatch_01VSDneWuSbf8mShA32jiWrX | 6 分 15 秒 | 790,370 | 326,193 | $3.63 | 已取代(措辭修訂後全體重跑) | 01-01-01-tag |
|
||||||
| 2026-08-05 | 01-02-01-merge | run1 | claude-opus-5 | msgbatch_01Wi9xEoQKCy1nZK6myVNg77 | 11 分 34 秒 | 63,368 | 70,175 | $2.07 | 已取代(驗證不過:漏 59、發明 140、重複 3) |
|
| 2026-08-05 | 1-tag | run1-rescue-288 | claude-sonnet-4-6 | msgbatch_019Hq6bNVXVmp4DjRZ2cgVda | 2 分 5 秒 | 1,015 | 376 | $0.01 | 已取代(措辭修訂後全體重跑,該首原生通過)| 01-01-01-tag |
|
||||||
| 2026-08-05 | 01-02-01-merge | run1 | claude-fable-5 | msgbatch_01M15hUs8P9FDTqWvAHdc1cA | 32 分 45 秒 | 63,368 | 107,721 | $3.01 | 現行歸檔(JSON 語法毀損,驗證不過) |
|
| 2026-08-05 | 1-tag | run1 | claude-sonnet-4-6 | msgbatch_01VgZ77KAPGWmuu3PnQFqZ7Q | 7 分 1 秒 | 794,913 | 326,435 | $3.64 | 現行 | 01-01-01-tag |
|
||||||
| 2026-08-05 | 01-02-01-merge | run1 | claude-fable-5 | msgbatch_012stRP28uuQDQwobm41HMLh | — | 0 | 0 | $0.00 | 拒收(thinking.type.enabled 不支援) |
|
| 2026-08-05 | 1-tag | run2 | claude-sonnet-4-6 | msgbatch_01TLFey3L4fimKxcebTZQGYn | 4 分 21 秒 | 794,913 | 328,324 | $3.65 | 現行 | 01-01-01-tag |
|
||||||
| 2026-08-05 | 01-02-01-merge | run1 | claude-fable-5 | msgbatch_01KpS7AMx1zAHTcNujgGsouJ | 26 分 28 秒 | 63,368 | 128,000 | $3.52 | 現行歸檔(effort max:128k 全耗於推理,正文空白) |
|
| 2026-08-05 | 已廢棄 | run1 | claude-sonnet-4-6 | msgbatch_01BvYMFmH8zrUq9SNxWSNba7 | 9 分 14 秒 | 46,454 | 60,974 | $0.53 | 已取代(完整分割驗證不過:漏 366、重複分派 511、撞名 4) | 01-02-01-merge |
|
||||||
| 2026-08-05 | 03-01-code | run1 | claude-sonnet-4-6 | msgbatch_01L7VepTxSNz5vruvzzjuL2c | 6 分 36 秒 | 1,247,264 | 492,730 | $5.57 | 36 首輸出遭內容過濾攔阻,待修訂重跑 |
|
| 2026-08-05 | 已廢棄 | run1 | claude-opus-5 | msgbatch_018TkYpdT3DsZqq2q94FQCUN | — | 0 | 0 | $0.00 | 拒收(temperature 已棄用) | 01-02-01-merge |
|
||||||
| 2026-08-05 | 03-01-code | 引述減量實驗(36 首)| claude-sonnet-4-6 | msgbatch_01Hskdhi2DkudYmgZhhgFts7 | 3 分 51 秒 | 56,458 | 11,074 | $0.17 | 實驗:每碼一行引述,36 首全數通過過濾;歸檔不入 repo |
|
| 2026-08-05 | 已廢棄 | run1 | claude-opus-4-8 | msgbatch_018Wuux4adz5mWSvzJjfgLxb | — | 0 | 0 | $0.00 | 拒收(temperature 已棄用) | 01-02-01-merge |
|
||||||
| 2026-08-05 | 03-01-code | run1 | claude-sonnet-4-6 | msgbatch_01GG5Ez9KT1pPwtQaY4sW7tv | 4 分 35 秒 | 1,293,724 | 269,166 | $3.96 | 過濾零攔阻;song-168、song-590 因餘額用盡失敗,song-775 拒答 |
|
| 2026-08-05 | 已廢棄 | run1 | claude-opus-4-6 | msgbatch_01Lmt2j1Txyof9CctUK7CbHA | 12 分 19 秒 | 46,454 | 54,864 | $0.80 | 已取代(圍欄違規;漏 190、發明 151、重複 11) | 01-02-01-merge |
|
||||||
| 2026-08-05 | 03-01-code | 101 碼樹狀探測(148 首)| claude-sonnet-4-6 | msgbatch_017jo3E5gWVks9b39iWMTqz3 | 2 小時 11 分 | 385,270 | 79,690 | $0.87 | 實驗:k=100 葉碼+women-power;零違規碼; 歸檔不入 repo |
|
| 2026-08-05 | 已廢棄 | run1 | claude-opus-5 | msgbatch_012SEfebTnjU5uWN6hhfnhhh | 8 分 45 秒 | 63,368 | 64,000 | $1.92 | 已取代(64k 截斷) | 01-02-01-merge |
|
||||||
| 2026-08-06 | 命名實驗(100 組)| — | claude-sonnet-4-6 | msgbatch_01XSi1YtWzdVyYUzYh7DQRWg | 3 分 5 秒 | 55,090 | 1,168 | $0.09 | 實驗:LLM 命名對照 medoid,未採用;歸檔不入 repo |
|
| 2026-08-05 | 已廢棄 | run1 | claude-opus-5 | msgbatch_01Wi9xEoQKCy1nZK6myVNg77 | 11 分 34 秒 | 63,368 | 70,175 | $2.07 | 已取代(驗證不過:漏 59、發明 140、重複 3) | 01-02-01-merge |
|
||||||
| 2026-08-06 | 命名實驗(100 組)| — | claude-fable-5 | msgbatch_01Y8SJj1h1ZuSQRErhqkvgZE | 3 分 2 秒 | 76,211 | 2,049 | $0.43 | 實驗:同上,加禁用 themes;未採用;歸檔不入 repo |
|
| 2026-08-05 | 已廢棄 | run1 | claude-fable-5 | msgbatch_01M15hUs8P9FDTqWvAHdc1cA | 32 分 45 秒 | 63,368 | 107,721 | $3.01 | 現行歸檔(JSON 語法毀損,驗證不過) | 01-02-01-merge |
|
||||||
| 2026-08-06 | 03-01-code | run1 | claude-sonnet-4-6 | msgbatch_01GS1opvurvsf62oknnQxhtx | 3 分 46 秒 | 1,625,458 | 362,759 | $5.16 | 現行(101 碼;883 首全數有效,零攔阻) |
|
| 2026-08-05 | 已廢棄 | run1 | claude-fable-5 | msgbatch_012stRP28uuQDQwobm41HMLh | — | 0 | 0 | $0.00 | 拒收(thinking.type.enabled 不支援) | 01-02-01-merge |
|
||||||
| 2026-08-06 | 03-01-code | run2 | claude-sonnet-4-6 | msgbatch_01CxnwNLWzZbRpZK7UAdpb8i | 5 分 22 秒 | 1,625,458 | 364,098 | $5.17 | 現行(101 碼;883 首全數有效,零攔阻) |
|
| 2026-08-05 | 已廢棄 | run1 | claude-fable-5 | msgbatch_01KpS7AMx1zAHTcNujgGsouJ | 26 分 28 秒 | 63,368 | 128,000 | $3.52 | 現行歸檔(effort max:128k 全耗於推理,正文空白) | 01-02-01-merge |
|
||||||
| 2026-08-06 | 03-02-arbitration | — | claude-sonnet-4-6 | msgbatch_019xcQXwrbwDGFc9nE5M8AjE | 4 分 0 秒 | 763,193 | 42,389 | $1.46 | 已取代(2 首遭內容過濾攔阻、13 首輸出夾帶散文;定義檔修訂後重跑) |
|
| 2026-08-05 | 3a-code | run1 | claude-sonnet-4-6 | msgbatch_01L7VepTxSNz5vruvzzjuL2c | 6 分 36 秒 | 1,247,264 | 492,730 | $5.57 | 36 首輸出遭內容過濾攔阻,待修訂重跑 | 03-01-code |
|
||||||
| 2026-08-06 | 03-02-arbitration | — | claude-sonnet-4-6 | msgbatch_01N7bDbXRSfAVUzzaj2thKeR | 4 分 20 秒 | 781,037 | 39,315 | $1.47 | 現行(644 首全數有效,零攔阻;保留 1,481/送裁 1,699) |
|
| 2026-08-05 | 3a-code | 引述減量實驗(36 首)| claude-sonnet-4-6 | msgbatch_01Hskdhi2DkudYmgZhhgFts7 | 3 分 51 秒 | 56,458 | 11,074 | $0.17 | 實驗:每碼一行引述,36 首全數通過過濾;歸檔不入 repo | 03-01-code |
|
||||||
| 2026-08-06 | 03-code | run3 | claude-sonnet-4-6 | msgbatch_01KnkCaGETnFJrPddrxZTYHA | 6 分 26 秒 | 1,625,458 | 363,840 | $5.17 | 現行(101 碼;883 首全數有效,零攔阻) |
|
| 2026-08-05 | 3a-code | run1 | claude-sonnet-4-6 | msgbatch_01GG5Ez9KT1pPwtQaY4sW7tv | 4 分 35 秒 | 1,293,724 | 269,166 | $3.96 | 過濾零攔阻;song-168、song-590 因餘額用盡失敗,song-775 拒答 | 03-01-code |
|
||||||
| 2026-08-14 | 04-group | run1 | claude-sonnet-4-6 | msgbatch_01UPNedog6feQzJ9WVfSAxBD | 1 分 1 秒 | 3,129 | 470 | $0.01 | 已取代(僅 3 群;改納 women-power 群後重跑) |
|
| 2026-08-05 | 3a-code | 101 碼樹狀探測(148 首)| claude-sonnet-4-6 | msgbatch_017jo3E5gWVks9b39iWMTqz3 | 2 小時 11 分 | 385,270 | 79,690 | $0.87 | 實驗:k=100 葉碼+women-power;零違規碼; 歸檔不入 repo | 03-01-code |
|
||||||
| 2026-08-14 | 04-group | run1 | claude-sonnet-4-6 | msgbatch_01KntJgxdicStaMjNL12P3zi | 1 分 25 秒 | 4,170 | 558 | $0.01 | 已取代(改以 claude-fable-5 執行;vulnerable 輸出含詞彙表外碼 1 筆;歸檔另行私人備份,不入版本庫) |
|
| 2026-08-06 | 命名實驗(100 組)| — | claude-sonnet-4-6 | msgbatch_01XSi1YtWzdVyYUzYh7DQRWg | 3 分 5 秒 | 55,090 | 1,168 | $0.09 | 實驗:LLM 命名對照 medoid,未採用;歸檔不入 repo | |
|
||||||
| 2026-08-14 | 04-group | run1 | claude-fable-5 | msgbatch_01Mr6goBb2Efa4YrqCprbP4U | 55 秒 | 5,553 | 2,049 | $0.08 | 現行(4 群;零違規碼;temperature 與 thinking 參數不適用於本模型,未送出) |
|
| 2026-08-06 | 命名實驗(100 組)| — | claude-fable-5 | msgbatch_01Y8SJj1h1ZuSQRErhqkvgZE | 3 分 2 秒 | 76,211 | 2,049 | $0.43 | 實驗:同上,加禁用 themes;未採用;歸檔不入 repo | |
|
||||||
| 2026-08-14 | 04-group | run2 | claude-fable-5 | msgbatch_01C17xW3YBefThTYZ83g7KkL | 1 分 21 秒 | 5,553 | 1,891 | $0.08 | 現行(4 群;零違規碼) |
|
| 2026-08-06 | 3a-code | run1 | claude-sonnet-4-6 | msgbatch_01GS1opvurvsf62oknnQxhtx | 3 分 46 秒 | 1,625,458 | 362,759 | $5.16 | 現行(101 碼;883 首全數有效,零攔阻) | 03-01-code |
|
||||||
| 2026-08-14 | 04-group | run3 | claude-fable-5 | msgbatch_01DveEMYyjCAYe6wxpcCD87V | 2 分 9 秒 | 5,553 | 1,975 | $0.08 | 現行(4 群;零違規碼) |
|
| 2026-08-06 | 3a-code | run2 | claude-sonnet-4-6 | msgbatch_01CxnwNLWzZbRpZK7UAdpb8i | 5 分 22 秒 | 1,625,458 | 364,098 | $5.17 | 現行(101 碼;883 首全數有效,零攔阻) | 03-01-code |
|
||||||
| 2026-08-15 | 05-01-read | run1 | claude-fable-5 | msgbatch_018NJejke75o7kzKyjZeeKwi | 2 分 12 秒 | 189,607 | 181,299 | $5.48 | 現行(144/145 有效;song-444 平台錯誤,單筆補送) |
|
| 2026-08-06 | 已廢棄 | — | claude-sonnet-4-6 | msgbatch_019xcQXwrbwDGFc9nE5M8AjE | 4 分 0 秒 | 763,193 | 42,389 | $1.46 | 已取代(2 首遭內容過濾攔阻、13 首輸出夾帶散文;定義檔修訂後重跑) | 03-02-arbitration |
|
||||||
| 2026-08-15 | 05-01-read | run1-rescue-444 | claude-fable-5 | msgbatch_01DgygMm6iYn2qEnyqHkTMfa | 1 分 52 秒 | 1,852 | 1,557 | $0.05 | 現行(run1 之 song-444 單筆補送,成功) |
|
| 2026-08-06 | 已廢棄 | — | claude-sonnet-4-6 | msgbatch_01N7bDbXRSfAVUzzaj2thKeR | 4 分 20 秒 | 781,037 | 39,315 | $1.47 | 現行(644 首全數有效,零攔阻;保留 1,481/送裁 1,699) | 03-02-arbitration |
|
||||||
| 2026-08-15 | 05-01-read | run2 | claude-fable-5 | msgbatch_011txWpWDtdkNXNiPA74ZUfQ | 15 分 25 秒 | 191,459 | 185,630 | $5.60 | 現行(145/145 有效) |
|
| 2026-08-06 | 3a-code | run3 | claude-sonnet-4-6 | msgbatch_01KnkCaGETnFJrPddrxZTYHA | 6 分 26 秒 | 1,625,458 | 363,840 | $5.17 | 現行(101 碼;883 首全數有效,零攔阻) | 03-code |
|
||||||
| 2026-08-15 | 05-01-read | run3 | claude-fable-5 | msgbatch_01Unn7C7WCnuVoLraJRXksoG | 7 分 59 秒 | 191,459 | 190,433 | $5.72 | 現行(145/145 有效) |
|
| 2026-08-14 | 4-group | run1 | claude-sonnet-4-6 | msgbatch_01UPNedog6feQzJ9WVfSAxBD | 1 分 1 秒 | 3,129 | 470 | $0.01 | 已取代(僅 3 群;改納 women-power 群後重跑) | 04-group |
|
||||||
| 2026-08-15 | 05-02-consolidate | run1 | claude-fable-5 | msgbatch_01WPyH4pCCiieHs4yU3xtHzN | 2 分 28 秒 | 326,287 | 242,785 | $7.70 | 現行(145/145 有效) |
|
| 2026-08-14 | 4-group | run1 | claude-sonnet-4-6 | msgbatch_01KntJgxdicStaMjNL12P3zi | 1 分 25 秒 | 4,170 | 558 | $0.01 | 已取代(改以 claude-fable-5 執行;vulnerable 輸出含詞彙表外碼 1 筆;歸檔備份於 excludes/experiments/) | 04-group |
|
||||||
| 2026-08-15 | 05-03-synthesize | run1 | claude-fable-5 | msgbatch_01XbNQb1SFrUxhk2ZWF7epPA | 3 分 0 秒 | 103,888 | 9,982 | $0.77 | 現行(15 個樣態;研究者審定用草稿) |
|
| 2026-08-14 | 4-group | run1 | claude-fable-5 | msgbatch_01Mr6goBb2Efa4YrqCprbP4U | 55 秒 | 5,553 | 2,049 | $0.08 | 現行(4 群;零違規碼;temperature 與 thinking 參數不適用於本模型,未送出) | 04-group |
|
||||||
| 2026-08-15 | 05-03-synthesize | run2 | claude-fable-5 | msgbatch_01FKqq8U6m5HLLsx7KpQES26 | 3 分 4 秒 | 13,573 | 6,883 | $0.24 | 現行(男聲群 14 首;13 個樣態) |
|
| 2026-08-14 | 4-group | run2 | claude-fable-5 | msgbatch_01C17xW3YBefThTYZ83g7KkL | 1 分 21 秒 | 5,553 | 1,891 | $0.08 | 現行(4 群;零違規碼) | 04-group |
|
||||||
| 2026-08-15 | 05-03-synthesize | run3 | claude-fable-5 | msgbatch_017tfhiRTnaxqCp6YFyoivsu | 4 分 4 秒 | 60,727 | 8,519 | $0.52 | 現行(女聲群 98 首;14 個樣態) |
|
| 2026-08-14 | 4-group | run3 | claude-fable-5 | msgbatch_01DveEMYyjCAYe6wxpcCD87V | 2 分 9 秒 | 5,553 | 1,975 | $0.08 | 現行(4 群;零違規碼) | 04-group |
|
||||||
| 2026-08-15 | 05-03-synthesize | run4 | claude-fable-5 | msgbatch_01SNjTLiNPEba4LW2tcUAcTJ | 4 分 3 秒 | 29,958 | 9,793 | $0.39 | 現行(混合群 30 首;16 個樣態) |
|
| 2026-08-15 | 5a-read | run1 | claude-fable-5 | msgbatch_018NJejke75o7kzKyjZeeKwi | 2 分 12 秒 | 189,607 | 181,299 | $5.48 | 現行(144/145 有效;song-444 平台錯誤,單筆補送) | 05-01-read |
|
||||||
| 2026-08-15 | 05-04-annotate | run1 | claude-fable-5 | msgbatch_01H9gXXhNbhZG6gzZtvMHxhh | 3 分 9 秒 | 815,492 | 300,901 | $11.60 | 現行(109/111 有效;song-177、song-199 遭 max_tokens 截斷,單筆補送) |
|
| 2026-08-15 | 5a-read | run1-rescue-444 | claude-fable-5 | msgbatch_01DgygMm6iYn2qEnyqHkTMfa | 1 分 52 秒 | 1,852 | 1,557 | $0.05 | 現行(run1 之 song-444 單筆補送,成功) | 05-01-read |
|
||||||
| 2026-08-15 | 05-04-annotate | run1-rescue-177-199 | claude-fable-5 | msgbatch_01YEdroZ51b8btPfG1aqPFCD | 3 分 5 秒 | 17,802 | 11,601 | $0.38 | 現行(run1 之 2 筆補送,max_tokens 提為 16000,成功) |
|
| 2026-08-15 | 5a-read | run2 | claude-fable-5 | msgbatch_011txWpWDtdkNXNiPA74ZUfQ | 15 分 25 秒 | 191,459 | 185,630 | $5.60 | 現行(145/145 有效) | 05-01-read |
|
||||||
| 2026-08-15 | 05-04-annotate | — | claude-fable-5 | msgbatch_01G67STqZsHpnhedat2HRU85 | 3 分 0 秒 | 17,802 | 14,162 | $0.44 | 誤送(補送批次輪詢中斷後誤判死亡而重送;原批次自行完成並歸檔為現行,本批次取消不及、兩筆皆完成,結果棄用) |
|
| 2026-08-15 | 5a-read | run3 | claude-fable-5 | msgbatch_01Unn7C7WCnuVoLraJRXksoG | 7 分 59 秒 | 191,459 | 190,433 | $5.72 | 現行(145/145 有效) | 05-01-read |
|
||||||
| 2026-08-15 | 05-04-annotate | run2 | claude-fable-5 | msgbatch_01VVJpsfidNtF66GnHQdWnGR | 9 分 16 秒 | 815,492 | 306,582 | $11.74 | 現行(111/111 有效;max_tokens 16000,零截斷) |
|
| 2026-08-15 | 5b-consolidate | run1 | claude-fable-5 | msgbatch_01WPyH4pCCiieHs4yU3xtHzN | 2 分 28 秒 | 326,287 | 242,785 | $7.70 | 現行(145/145 有效) | 05-02-consolidate |
|
||||||
| 2026-08-15 | 05-04-annotate | run3 | claude-fable-5 | msgbatch_01AsYvZd8rsTYVfUYLzWfeMK | 4 分 12 秒 | 815,492 | 298,512 | $11.54 | 現行(111/111 有效;max_tokens 16000,零截斷) |
|
| 2026-08-15 | 5c-synthesize | run1 | claude-fable-5 | msgbatch_01XbNQb1SFrUxhk2ZWF7epPA | 3 分 0 秒 | 103,888 | 9,982 | $0.77 | 現行(15 個樣態;研究者審定用草稿) | 05-03-synthesize |
|
||||||
|
| 2026-08-15 | 5c-synthesize | run2 | claude-fable-5 | msgbatch_01FKqq8U6m5HLLsx7KpQES26 | 3 分 4 秒 | 13,573 | 6,883 | $0.24 | 現行(男聲群 14 首;13 個樣態) | 05-03-synthesize |
|
||||||
|
| 2026-08-15 | 5c-synthesize | run3 | claude-fable-5 | msgbatch_017tfhiRTnaxqCp6YFyoivsu | 4 分 4 秒 | 60,727 | 8,519 | $0.52 | 現行(女聲群 98 首;14 個樣態) | 05-03-synthesize |
|
||||||
|
| 2026-08-15 | 5c-synthesize | run4 | claude-fable-5 | msgbatch_01SNjTLiNPEba4LW2tcUAcTJ | 4 分 3 秒 | 29,958 | 9,793 | $0.39 | 現行(混合群 30 首;16 個樣態) | 05-03-synthesize |
|
||||||
|
| 2026-08-15 | 5d-annotate | run1 | claude-fable-5 | msgbatch_01H9gXXhNbhZG6gzZtvMHxhh | 3 分 9 秒 | 815,492 | 300,901 | $11.60 | 現行(109/111 有效;song-177、song-199 遭 max_tokens 截斷,單筆補送) | 05-04-annotate |
|
||||||
|
| 2026-08-15 | 5d-annotate | run1-rescue-177-199 | claude-fable-5 | msgbatch_01YEdroZ51b8btPfG1aqPFCD | 3 分 5 秒 | 17,802 | 11,601 | $0.38 | 現行(run1 之 2 筆補送,max_tokens 提為 16000,成功) | 05-04-annotate |
|
||||||
|
| 2026-08-15 | 5d-annotate | — | claude-fable-5 | msgbatch_01G67STqZsHpnhedat2HRU85 | 3 分 0 秒 | 17,802 | 14,162 | $0.44 | 誤送(補送批次輪詢中斷後誤判死亡而重送;原批次自行完成並歸檔為現行,本批次取消不及、兩筆皆完成,結果棄用) | 05-04-annotate |
|
||||||
|
| 2026-08-15 | 5d-annotate | run2 | claude-fable-5 | msgbatch_01VVJpsfidNtF66GnHQdWnGR | 9 分 16 秒 | 815,492 | 306,582 | $11.74 | 現行(111/111 有效;max_tokens 16000,零截斷) | 05-04-annotate |
|
||||||
|
| 2026-08-15 | 5d-annotate | run3 | claude-fable-5 | msgbatch_01AsYvZd8rsTYVfUYLzWfeMK | 4 分 12 秒 | 815,492 | 298,512 | $11.54 | 現行(111/111 有效;max_tokens 16000,零截斷) | 05-04-annotate |
|
||||||
|
|
||||||
累計支出:$125.65。
|
累計支出:$125.65。
|
||||||
|
|||||||
File diff suppressed because it is too large
Load Diff
@@ -24,8 +24,7 @@ keyword set for ``export-llm-input --extras`` is written as a JSON
|
|||||||
file holding the group name keywords plus every extra a-priori
|
file holding the group name keywords plus every extra a-priori
|
||||||
keyword the caller gives with the repeatable ``--extra-keyword``
|
keyword the caller gives with the repeatable ``--extra-keyword``
|
||||||
command-line option, as
|
command-line option, as
|
||||||
:attr:`KeywordsToMerge.KEYWORDS_TO_MERGE_JSON`; with no
|
:attr:`KeywordsToMerge.KEYWORDS_TO_MERGE_JSON`. No default
|
||||||
``--extra-keyword``, it holds the group names alone. No default
|
|
||||||
extra keyword is ever injected; the caller supplies each one
|
extra keyword is ever injected; the caller supplies each one
|
||||||
consciously. Finally, the command-line choices and the
|
consciously. Finally, the command-line choices and the
|
||||||
environment that produced the numbers -- neither recoverable from
|
environment that produced the numbers -- neither recoverable from
|
||||||
@@ -160,15 +159,8 @@ class KeywordPooler:
|
|||||||
if line.strip() == "":
|
if line.strip() == "":
|
||||||
continue
|
continue
|
||||||
record: Any = json.loads(line)
|
record: Any = json.loads(line)
|
||||||
if not isinstance(record, dict) or "id" not in record:
|
|
||||||
raise ValueError(
|
|
||||||
f"{path}: record without \"id\": {line}")
|
|
||||||
if "error" in record:
|
if "error" in record:
|
||||||
continue
|
continue
|
||||||
if "text" not in record:
|
|
||||||
raise ValueError(
|
|
||||||
f"{path}: id {record['id']}: record without"
|
|
||||||
" \"text\" or \"error\"")
|
|
||||||
song_id: int = cls.__parse_song_id(record["id"], path)
|
song_id: int = cls.__parse_song_id(record["id"], path)
|
||||||
try:
|
try:
|
||||||
keywords: Any = json.loads(
|
keywords: Any = json.loads(
|
||||||
@@ -297,7 +289,8 @@ class KeywordGroups:
|
|||||||
class KeywordClusterer:
|
class KeywordClusterer:
|
||||||
"""The clusterer of the pooled keywords into coding groups."""
|
"""The clusterer of the pooled keywords into coding groups."""
|
||||||
|
|
||||||
DEFAULT_MODEL: str = "sentence-transformers/all-mpnet-base-v2"
|
DEFAULT_MODEL: ClassVar[str] \
|
||||||
|
= "sentence-transformers/all-mpnet-base-v2"
|
||||||
"""The sentence embedding model used when the caller names
|
"""The sentence embedding model used when the caller names
|
||||||
none."""
|
none."""
|
||||||
|
|
||||||
@@ -668,11 +661,7 @@ def parse_args(argv: list[str] | None) -> argparse.Namespace:
|
|||||||
parser.add_argument(
|
parser.add_argument(
|
||||||
"output_dir", type=Path,
|
"output_dir", type=Path,
|
||||||
help="the output directory, created if missing, that"
|
help="the output directory, created if missing, that"
|
||||||
f" receives {PooledKeywords.SOURCE_KEYWORDS_TXT},"
|
" receives the run's output artifacts")
|
||||||
f" {KeywordGroups.RESULT_KEYWORDS_TXT},"
|
|
||||||
f" {KeywordGroups.RESULT_GROUPS_CSV},"
|
|
||||||
f" {KeywordsToMerge.KEYWORDS_TO_MERGE_JSON},"
|
|
||||||
f" and {RunMeta.META_JSON}")
|
|
||||||
parser.add_argument(
|
parser.add_argument(
|
||||||
"--model", default=model,
|
"--model", default=model,
|
||||||
help=f"the sentence embedding model (default \"{model}\")")
|
help=f"the sentence embedding model (default \"{model}\")")
|
||||||
@@ -698,16 +687,9 @@ def parse_args(argv: list[str] | None) -> argparse.Namespace:
|
|||||||
def main(argv: list[str] | None = None) -> int:
|
def main(argv: list[str] | None = None) -> int:
|
||||||
"""Pool the two tagging runs' keywords and cluster them.
|
"""Pool the two tagging runs' keywords and cluster them.
|
||||||
|
|
||||||
Writes the five fixed-named artifacts under the output
|
Creates the output directory (with parents) if it does not
|
||||||
directory, creating it (with parents) if it does not exist:
|
exist. Each output artifact is written as soon as its
|
||||||
the pooled keyword text file; then the group membership CSV
|
content is computed, so when the input is rejected, or an
|
||||||
file, holding the clustering result alone; the group name
|
|
||||||
keyword text file, holding the same group names as a readable
|
|
||||||
list; the coding keyword set JSON file, holding the group
|
|
||||||
names plus every extra keyword given via ``--extra-keyword``;
|
|
||||||
and the run metadata JSON file, recording the command-line
|
|
||||||
choices and the environment. Each file is written as soon as
|
|
||||||
its content is computed, so when the input is rejected, or an
|
|
||||||
extra keyword duplicates a group name or another extra
|
extra keyword duplicates a group name or another extra
|
||||||
keyword, the output directory holds whatever the steps before
|
keyword, the output directory holds whatever the steps before
|
||||||
the failing one produced, and the error message names what
|
the failing one produced, and the error message names what
|
||||||
@@ -719,8 +701,8 @@ def main(argv: list[str] | None = None) -> int:
|
|||||||
"""
|
"""
|
||||||
started: float = time.monotonic()
|
started: float = time.monotonic()
|
||||||
args: argparse.Namespace = parse_args(argv)
|
args: argparse.Namespace = parse_args(argv)
|
||||||
args.output_dir.mkdir(parents=True, exist_ok=True)
|
|
||||||
try:
|
try:
|
||||||
|
args.output_dir.mkdir(parents=True, exist_ok=True)
|
||||||
source: PooledKeywords = KeywordPooler(
|
source: PooledKeywords = KeywordPooler(
|
||||||
args.run_dir_1, args.run_dir_2, args.output_dir).run()
|
args.run_dir_1, args.run_dir_2, args.output_dir).run()
|
||||||
clusters: KeywordGroups = KeywordClusterer(
|
clusters: KeywordGroups = KeywordClusterer(
|
||||||
@@ -735,7 +717,7 @@ def main(argv: list[str] | None = None) -> int:
|
|||||||
f"Done. Clustered {len(source.keywords)} keywords into"
|
f"Done. Clustered {len(source.keywords)} keywords into"
|
||||||
f" {len(clusters.names)}. {elapsed} elapsed.",
|
f" {len(clusters.names)}. {elapsed} elapsed.",
|
||||||
file=sys.stderr)
|
file=sys.stderr)
|
||||||
except ClusterError as error:
|
except (ClusterError, OSError) as error:
|
||||||
print(f"error: {error}", file=sys.stderr)
|
print(f"error: {error}", file=sys.stderr)
|
||||||
return 1
|
return 1
|
||||||
return 0
|
return 0
|
||||||
|
|||||||
@@ -11,26 +11,17 @@ project's lyrics-only firewall: the output carries only the
|
|||||||
lyrics text of each song, identified by an opaque song key; no
|
lyrics text of each song, identified by an opaque song key; no
|
||||||
title, artist, or chart data crosses into the LLM input.
|
title, artist, or chart data crosses into the LLM input.
|
||||||
|
|
||||||
With ``--extras``, each record's ``content`` becomes a JSON
|
With ``--extras`` and ``--extras-per-id``, a record's content may
|
||||||
object serialized as a string, its ``lyrics`` key holding the
|
carry extra parameters alongside the lyrics, so a step that needs
|
||||||
song's lyrics followed by the keys of the given extras file in
|
them can get them without this module knowing what they mean; see
|
||||||
their file order, so a step that needs parameters alongside the
|
the exporter's content-building step for how the two merge.
|
||||||
lyrics can carry them without this module knowing what they mean.
|
|
||||||
|
|
||||||
With ``--extras-per-id``, the same merge happens per song: the
|
|
||||||
given file maps a song ID to the extra keys of that one song, and
|
|
||||||
the export is restricted to the song IDs the file names, so a step
|
|
||||||
that revisits only some of the songs, each with its own parameters,
|
|
||||||
gets exactly those records. The two options may be given together,
|
|
||||||
in which case a record's keys are ``lyrics``, the shared extras'
|
|
||||||
keys, then that song's own keys, each group in its file order.
|
|
||||||
"""
|
"""
|
||||||
import argparse
|
import argparse
|
||||||
import json
|
import json
|
||||||
import sys
|
import sys
|
||||||
import time
|
import time
|
||||||
from pathlib import Path
|
from pathlib import Path
|
||||||
from typing import Any
|
from typing import Any, ClassVar
|
||||||
|
|
||||||
import sqlalchemy as sa
|
import sqlalchemy as sa
|
||||||
from sqlalchemy.orm import Session
|
from sqlalchemy.orm import Session
|
||||||
@@ -40,6 +31,270 @@ from ..models import Song
|
|||||||
from ..utils import format_duration
|
from ..utils import format_duration
|
||||||
|
|
||||||
|
|
||||||
|
class LlmInputExporter:
|
||||||
|
"""The exporter of the LLM input JSONL file."""
|
||||||
|
|
||||||
|
__LYRICS_KEY: ClassVar[str] = "lyrics"
|
||||||
|
"""The key holding the lyrics in a record's merged content,
|
||||||
|
and the key forbidden in an extras file."""
|
||||||
|
|
||||||
|
def __init__(
|
||||||
|
self, output_jsonl: Path, extras: Path | None = None,
|
||||||
|
extras_per_id: Path | None = None) -> None:
|
||||||
|
"""Set up the exporter of the LLM input JSONL file.
|
||||||
|
|
||||||
|
:param output_jsonl: The JSONL output file.
|
||||||
|
:param extras: The extras JSON file, or None for none.
|
||||||
|
:param extras_per_id: The per-ID extras JSON file, or
|
||||||
|
None for none.
|
||||||
|
"""
|
||||||
|
self.__output_jsonl: Path = output_jsonl
|
||||||
|
"""The JSONL output file."""
|
||||||
|
self.__extras_path: Path | None = extras
|
||||||
|
"""The extras JSON file, or None for none."""
|
||||||
|
self.__extras_per_id_path: Path | None = extras_per_id
|
||||||
|
"""The per-ID extras JSON file, or None for none."""
|
||||||
|
|
||||||
|
def run(self) -> int:
|
||||||
|
"""Export the songs' lyrics to the output JSONL file.
|
||||||
|
|
||||||
|
:return: The number of songs exported.
|
||||||
|
:raises OSError: When a file cannot be read or written.
|
||||||
|
:raises sqlalchemy.exc.SQLAlchemyError: When the working
|
||||||
|
store cannot be read.
|
||||||
|
:raises ValueError: When an extras file is malformed, an
|
||||||
|
exported song has no lyrics, or the per-ID extras
|
||||||
|
name a song the working store does not have.
|
||||||
|
"""
|
||||||
|
session: Session = ds.get_db()
|
||||||
|
try:
|
||||||
|
extras: dict[str, Any] | None = None
|
||||||
|
if self.__extras_path is not None:
|
||||||
|
extras = self.__load_extras(self.__extras_path)
|
||||||
|
extras_per_id: dict[str, dict[str, Any]] | None = None
|
||||||
|
if self.__extras_per_id_path is not None:
|
||||||
|
extras_per_id = self.__load_extras_per_id(
|
||||||
|
self.__extras_per_id_path)
|
||||||
|
lines: list[str] = self.__build_lines(
|
||||||
|
session, extras, extras_per_id)
|
||||||
|
finally:
|
||||||
|
session.close()
|
||||||
|
self.__write_output(lines)
|
||||||
|
return len(lines)
|
||||||
|
|
||||||
|
@staticmethod
|
||||||
|
def __no_duplicate_keys(
|
||||||
|
pairs: list[tuple[str, Any]]) -> dict[str, Any]:
|
||||||
|
"""Build a dict from JSON object pairs, rejecting
|
||||||
|
duplicates.
|
||||||
|
|
||||||
|
:param pairs: The key-value pairs of a JSON object, in
|
||||||
|
file order.
|
||||||
|
:return: The pairs as a dict, in file order.
|
||||||
|
:raises ValueError: When a key appears more than once.
|
||||||
|
"""
|
||||||
|
result: dict[str, Any] = {}
|
||||||
|
key: str
|
||||||
|
value: Any
|
||||||
|
for key, value in pairs:
|
||||||
|
if key in result:
|
||||||
|
raise ValueError(
|
||||||
|
f"duplicate key \"{key}\" in extras")
|
||||||
|
result[key] = value
|
||||||
|
return result
|
||||||
|
|
||||||
|
@classmethod
|
||||||
|
def __load_json_object(
|
||||||
|
cls, path: Path, label: str) -> dict[str, Any]:
|
||||||
|
"""Load a single JSON object from a file, in file order.
|
||||||
|
|
||||||
|
:param path: The JSON file.
|
||||||
|
:param label: The kind of file, for the error messages.
|
||||||
|
:return: The object, in file order.
|
||||||
|
:raises OSError: When the file cannot be read.
|
||||||
|
:raises ValueError: When the file is not valid JSON, is
|
||||||
|
not a JSON object, or has duplicate keys.
|
||||||
|
"""
|
||||||
|
with open(path, encoding="utf-8") as file:
|
||||||
|
text: str = file.read()
|
||||||
|
try:
|
||||||
|
data: Any = json.loads(
|
||||||
|
text, object_pairs_hook=cls.__no_duplicate_keys)
|
||||||
|
except json.JSONDecodeError as error:
|
||||||
|
raise ValueError(
|
||||||
|
f"invalid JSON in {label} file {path}: {error}") \
|
||||||
|
from error
|
||||||
|
if not isinstance(data, dict):
|
||||||
|
raise ValueError(
|
||||||
|
f"{label} file {path} must contain a JSON object")
|
||||||
|
return data
|
||||||
|
|
||||||
|
@classmethod
|
||||||
|
def __load_extras(cls, path: Path) -> dict[str, Any]:
|
||||||
|
"""Load the extras object from a JSON file.
|
||||||
|
|
||||||
|
:param path: The extras JSON file.
|
||||||
|
:return: The extras, in file order.
|
||||||
|
:raises OSError: When the file cannot be read.
|
||||||
|
:raises ValueError: When the file is not valid JSON, is
|
||||||
|
not a JSON object, has duplicate keys, or has a
|
||||||
|
"lyrics" key.
|
||||||
|
"""
|
||||||
|
data: dict[str, Any] = cls.__load_json_object(
|
||||||
|
path, "extras")
|
||||||
|
if cls.__LYRICS_KEY in data:
|
||||||
|
raise ValueError(
|
||||||
|
f"extras file {path} must not have a"
|
||||||
|
f" \"{cls.__LYRICS_KEY}\" key")
|
||||||
|
return data
|
||||||
|
|
||||||
|
@classmethod
|
||||||
|
def __load_extras_per_id(
|
||||||
|
cls, path: Path) -> dict[str, dict[str, Any]]:
|
||||||
|
"""Load the per-ID extras object from a JSON file.
|
||||||
|
|
||||||
|
:param path: The per-ID extras JSON file, mapping a song
|
||||||
|
ID, as ``song-<N>``, to the extras of that one song.
|
||||||
|
:return: The extras of each song ID, in file order, every
|
||||||
|
song's own extras in their file order too.
|
||||||
|
:raises OSError: When the file cannot be read.
|
||||||
|
:raises ValueError: When the file is not valid JSON, is
|
||||||
|
not a JSON object, has duplicate keys, has a song
|
||||||
|
whose value is not a JSON object, or has a song with
|
||||||
|
a "lyrics" key.
|
||||||
|
"""
|
||||||
|
data: dict[str, Any] = cls.__load_json_object(
|
||||||
|
path, "per-ID extras")
|
||||||
|
song_id: str
|
||||||
|
extras: Any
|
||||||
|
for song_id, extras in data.items():
|
||||||
|
if not isinstance(extras, dict):
|
||||||
|
raise ValueError(
|
||||||
|
f"per-ID extras file {path}: id {song_id}"
|
||||||
|
" must have a JSON object")
|
||||||
|
if cls.__LYRICS_KEY in extras:
|
||||||
|
raise ValueError(
|
||||||
|
f"per-ID extras file {path}: id {song_id}"
|
||||||
|
f" must not have a \"{cls.__LYRICS_KEY}\""
|
||||||
|
" key")
|
||||||
|
return data
|
||||||
|
|
||||||
|
def __build_lines(
|
||||||
|
self, session: Session,
|
||||||
|
extras: dict[str, Any] | None = None,
|
||||||
|
extras_per_id: dict[str, dict[str, Any]] | None
|
||||||
|
= None) -> list[str]:
|
||||||
|
"""Build the JSONL lines of the exported songs' lyrics.
|
||||||
|
|
||||||
|
Every song is exported, unless per-ID extras are given,
|
||||||
|
in which case only the songs they name are; see
|
||||||
|
:meth:`__build_content` for how the extras merge into a
|
||||||
|
record's content.
|
||||||
|
|
||||||
|
:param session: The database session.
|
||||||
|
:param extras: The extra parameters merged into every
|
||||||
|
record's content alongside the lyrics, in the order
|
||||||
|
they are to appear, or None for none.
|
||||||
|
:param extras_per_id: The extra parameters merged into
|
||||||
|
the content of one record alone, keyed by that
|
||||||
|
record's song ID and in the order they are to appear,
|
||||||
|
restricting the export to the song IDs they name, or
|
||||||
|
None for no such extras and no such restriction.
|
||||||
|
:return: The JSON lines, one per exported song, ordered by
|
||||||
|
song ID.
|
||||||
|
:raises ValueError: When an exported song has no lyrics,
|
||||||
|
or the per-ID extras name a song the working store
|
||||||
|
does not have.
|
||||||
|
"""
|
||||||
|
lines: list[str] = []
|
||||||
|
exported: set[str] = set()
|
||||||
|
song: Song
|
||||||
|
for song in session.scalars(
|
||||||
|
sa.select(Song).order_by(Song.id)):
|
||||||
|
song_id: str = f"song-{song.id}"
|
||||||
|
if extras_per_id is not None \
|
||||||
|
and song_id not in extras_per_id:
|
||||||
|
continue
|
||||||
|
if song.lyrics is None:
|
||||||
|
raise ValueError(
|
||||||
|
f"song {song.id} \"{song.title}\": no lyrics")
|
||||||
|
song_extras: dict[str, Any] | None = None \
|
||||||
|
if extras_per_id is None \
|
||||||
|
else extras_per_id[song_id]
|
||||||
|
content: str = self.__build_content(
|
||||||
|
song.lyrics, extras, song_extras)
|
||||||
|
record: dict[str, str] = {
|
||||||
|
"id": song_id, "content": content}
|
||||||
|
lines.append(json.dumps(record, ensure_ascii=False))
|
||||||
|
exported.add(song_id)
|
||||||
|
if extras_per_id is not None:
|
||||||
|
missing: list[str] = sorted(
|
||||||
|
set(extras_per_id) - exported)
|
||||||
|
if len(missing) > 0:
|
||||||
|
raise ValueError(
|
||||||
|
"the per-ID extras name songs the working"
|
||||||
|
f" store does not have: {', '.join(missing)}")
|
||||||
|
return lines
|
||||||
|
|
||||||
|
@classmethod
|
||||||
|
def __build_content(
|
||||||
|
cls, lyrics: str, extras: dict[str, Any] | None,
|
||||||
|
song_extras: dict[str, Any] | None) -> str:
|
||||||
|
"""Build the content of one exported record.
|
||||||
|
|
||||||
|
Without extras of either kind, a record's content is the
|
||||||
|
bare lyrics string. With ``--extras``, the content
|
||||||
|
becomes a JSON object serialized as a string, its
|
||||||
|
"lyrics" key holding the song's lyrics followed by the
|
||||||
|
keys of the given extras file, in their file order. With
|
||||||
|
``--extras-per-id``, the same merge happens per song: the
|
||||||
|
song's own extra keys follow the lyrics instead. When
|
||||||
|
both are given, a record's keys are "lyrics", the shared
|
||||||
|
extras' keys, then that song's own keys, each group in
|
||||||
|
its file order.
|
||||||
|
|
||||||
|
:param lyrics: The lyrics of the song.
|
||||||
|
:param extras: The extra parameters shared by every
|
||||||
|
record, in the order they are to appear, or None for
|
||||||
|
none.
|
||||||
|
:param song_extras: The extra parameters of this record
|
||||||
|
alone, in the order they are to appear, or None for
|
||||||
|
none.
|
||||||
|
:return: The bare lyrics when there are no extras of
|
||||||
|
either kind, or otherwise a JSON object serialized as
|
||||||
|
a string, whose first key is "lyrics" holding the
|
||||||
|
lyrics, followed by the shared extras' keys and then
|
||||||
|
this record's own keys, each group in its given
|
||||||
|
order.
|
||||||
|
"""
|
||||||
|
if extras is None and song_extras is None:
|
||||||
|
return lyrics
|
||||||
|
payload: dict[str, Any] = {cls.__LYRICS_KEY: lyrics}
|
||||||
|
if extras is not None:
|
||||||
|
payload.update(extras)
|
||||||
|
if song_extras is not None:
|
||||||
|
payload.update(song_extras)
|
||||||
|
return json.dumps(payload, ensure_ascii=False)
|
||||||
|
|
||||||
|
def __write_output(self, lines: list[str]) -> None:
|
||||||
|
"""Write the exported lines to the output JSONL file.
|
||||||
|
|
||||||
|
Creates the parent directory when it does not exist.
|
||||||
|
|
||||||
|
:param lines: The JSONL lines, in the output order.
|
||||||
|
:return: None.
|
||||||
|
:raises OSError: When the file cannot be written.
|
||||||
|
"""
|
||||||
|
self.__output_jsonl.parent.mkdir(
|
||||||
|
parents=True, exist_ok=True)
|
||||||
|
with open(
|
||||||
|
self.__output_jsonl, "w",
|
||||||
|
encoding="utf-8") as file:
|
||||||
|
line: str
|
||||||
|
for line in lines:
|
||||||
|
file.write(line + "\n")
|
||||||
|
|
||||||
|
|
||||||
def parse_args(argv: list[str] | None) -> argparse.Namespace:
|
def parse_args(argv: list[str] | None) -> argparse.Namespace:
|
||||||
"""Parse the command-line arguments.
|
"""Parse the command-line arguments.
|
||||||
|
|
||||||
@@ -56,227 +311,33 @@ def parse_args(argv: list[str] | None) -> argparse.Namespace:
|
|||||||
parser.add_argument(
|
parser.add_argument(
|
||||||
"--extras", type=Path, default=None,
|
"--extras", type=Path, default=None,
|
||||||
help="a JSON file holding a single JSON object of extra"
|
help="a JSON file holding a single JSON object of extra"
|
||||||
" parameters; when given, each record's \"content\""
|
" parameters merged into every record's content")
|
||||||
" becomes a JSON object string with a \"lyrics\" key"
|
|
||||||
" followed by the extras' keys, instead of the bare"
|
|
||||||
" lyrics string")
|
|
||||||
parser.add_argument(
|
parser.add_argument(
|
||||||
"--extras-per-id", type=Path, default=None,
|
"--extras-per-id", type=Path, default=None,
|
||||||
help="a JSON file holding a single JSON object that maps a"
|
help="a JSON file holding a single JSON object that maps a"
|
||||||
" song ID, as \"song-<N>\", to a JSON object of extra"
|
" song ID, as \"song-<N>\", to a JSON object of extra"
|
||||||
" parameters for that one song; the song's object is"
|
" parameters for that one song, restricting the"
|
||||||
" merged into its \"content\" the same way as with"
|
" export to the song IDs the file names")
|
||||||
" --extras, and the export is restricted to the song"
|
|
||||||
" IDs the file names")
|
|
||||||
return parser.parse_args(argv)
|
return parser.parse_args(argv)
|
||||||
|
|
||||||
|
|
||||||
def __no_duplicate_keys(
|
|
||||||
pairs: list[tuple[str, Any]]) -> dict[str, Any]:
|
|
||||||
"""Build a dict from JSON object pairs, rejecting duplicates.
|
|
||||||
|
|
||||||
:param pairs: The key-value pairs of a JSON object, in file
|
|
||||||
order.
|
|
||||||
:return: The pairs as a dict, in file order.
|
|
||||||
:raises ValueError: When a key appears more than once.
|
|
||||||
"""
|
|
||||||
result: dict[str, Any] = {}
|
|
||||||
key: str
|
|
||||||
value: Any
|
|
||||||
for key, value in pairs:
|
|
||||||
if key in result:
|
|
||||||
raise ValueError(f"duplicate key \"{key}\" in extras")
|
|
||||||
result[key] = value
|
|
||||||
return result
|
|
||||||
|
|
||||||
|
|
||||||
def __load_json_object(path: Path, label: str) -> dict[str, Any]:
|
|
||||||
"""Load a single JSON object from a file, in file order.
|
|
||||||
|
|
||||||
:param path: The JSON file.
|
|
||||||
:param label: The kind of file, for the error messages.
|
|
||||||
:return: The object, in file order.
|
|
||||||
:raises OSError: When the file cannot be read.
|
|
||||||
:raises ValueError: When the file is not valid JSON, is not
|
|
||||||
a JSON object, or has duplicate keys.
|
|
||||||
"""
|
|
||||||
with open(path, encoding="utf-8") as file:
|
|
||||||
text: str = file.read()
|
|
||||||
try:
|
|
||||||
data: Any = json.loads(
|
|
||||||
text, object_pairs_hook=__no_duplicate_keys)
|
|
||||||
except json.JSONDecodeError as error:
|
|
||||||
raise ValueError(
|
|
||||||
f"invalid JSON in {label} file {path}: {error}") \
|
|
||||||
from error
|
|
||||||
if not isinstance(data, dict):
|
|
||||||
raise ValueError(
|
|
||||||
f"{label} file {path} must contain a JSON object")
|
|
||||||
return data
|
|
||||||
|
|
||||||
|
|
||||||
def load_extras(path: Path) -> dict[str, Any]:
|
|
||||||
"""Load the extras object from a JSON file.
|
|
||||||
|
|
||||||
:param path: The extras JSON file.
|
|
||||||
:return: The extras, in file order.
|
|
||||||
:raises OSError: When the file cannot be read.
|
|
||||||
:raises ValueError: When the file is not valid JSON, is not
|
|
||||||
a JSON object, has duplicate keys, or has a "lyrics" key.
|
|
||||||
"""
|
|
||||||
data: dict[str, Any] = __load_json_object(path, "extras")
|
|
||||||
if "lyrics" in data:
|
|
||||||
raise ValueError(
|
|
||||||
f"extras file {path} must not have a \"lyrics\" key")
|
|
||||||
return data
|
|
||||||
|
|
||||||
|
|
||||||
def load_extras_per_id(path: Path) -> dict[str, dict[str, Any]]:
|
|
||||||
"""Load the per-ID extras object from a JSON file.
|
|
||||||
|
|
||||||
:param path: The per-ID extras JSON file, mapping a song ID,
|
|
||||||
as ``song-<N>``, to the extras of that one song.
|
|
||||||
:return: The extras of each song ID, in file order, every
|
|
||||||
song's own extras in their file order too.
|
|
||||||
:raises OSError: When the file cannot be read.
|
|
||||||
:raises ValueError: When the file is not valid JSON, is not
|
|
||||||
a JSON object, has duplicate keys, has a song whose value
|
|
||||||
is not a JSON object, or has a song with a "lyrics" key.
|
|
||||||
"""
|
|
||||||
data: dict[str, Any] = __load_json_object(path, "per-ID extras")
|
|
||||||
song_id: str
|
|
||||||
extras: Any
|
|
||||||
for song_id, extras in data.items():
|
|
||||||
if not isinstance(extras, dict):
|
|
||||||
raise ValueError(
|
|
||||||
f"per-ID extras file {path}: id {song_id} must"
|
|
||||||
" have a JSON object")
|
|
||||||
if "lyrics" in extras:
|
|
||||||
raise ValueError(
|
|
||||||
f"per-ID extras file {path}: id {song_id} must not"
|
|
||||||
" have a \"lyrics\" key")
|
|
||||||
return data
|
|
||||||
|
|
||||||
|
|
||||||
def __build_content(
|
|
||||||
lyrics: str,
|
|
||||||
extras: dict[str, Any] | None,
|
|
||||||
song_extras: dict[str, Any] | None) -> str:
|
|
||||||
"""Build the content of one exported record.
|
|
||||||
|
|
||||||
:param lyrics: The lyrics of the song.
|
|
||||||
:param extras: The extra parameters shared by every record,
|
|
||||||
in the order they are to appear, or None for none.
|
|
||||||
:param song_extras: The extra parameters of this record
|
|
||||||
alone, in the order they are to appear, or None for none.
|
|
||||||
:return: The bare lyrics when there are no extras of either
|
|
||||||
kind, or otherwise a JSON object serialized as a string,
|
|
||||||
whose first key is ``"lyrics"`` holding the lyrics,
|
|
||||||
followed by the shared extras' keys and then this
|
|
||||||
record's own keys, each group in its given order.
|
|
||||||
"""
|
|
||||||
if extras is None and song_extras is None:
|
|
||||||
return lyrics
|
|
||||||
payload: dict[str, Any] = {"lyrics": lyrics}
|
|
||||||
if extras is not None:
|
|
||||||
payload.update(extras)
|
|
||||||
if song_extras is not None:
|
|
||||||
payload.update(song_extras)
|
|
||||||
return json.dumps(payload, ensure_ascii=False)
|
|
||||||
|
|
||||||
|
|
||||||
def build_lines(
|
|
||||||
session: Session,
|
|
||||||
extras: dict[str, Any] | None = None,
|
|
||||||
extras_per_id: dict[str, dict[str, Any]] | None = None) \
|
|
||||||
-> list[str]:
|
|
||||||
"""Build the JSONL lines of the exported songs' lyrics.
|
|
||||||
|
|
||||||
Without extras of either kind, each record's ``content`` is
|
|
||||||
the bare lyrics string. With extras, ``content`` is a JSON
|
|
||||||
object serialized as a string, whose first key is ``"lyrics"``
|
|
||||||
holding the lyrics string, followed by the shared extras' keys
|
|
||||||
and then the song's own per-ID extras' keys, each group in its
|
|
||||||
given order.
|
|
||||||
|
|
||||||
Every song is exported, unless per-ID extras are given, in
|
|
||||||
which case only the songs they name are.
|
|
||||||
|
|
||||||
:param session: The database session.
|
|
||||||
:param extras: The extra parameters merged into every
|
|
||||||
record's content alongside the lyrics, in the order they
|
|
||||||
are to appear, or None for none.
|
|
||||||
:param extras_per_id: The extra parameters merged into the
|
|
||||||
content of one record alone, keyed by that record's song
|
|
||||||
ID and in the order they are to appear, restricting the
|
|
||||||
export to the song IDs they name, or None for no such
|
|
||||||
extras and no such restriction.
|
|
||||||
:return: The JSON lines, one per exported song, ordered by
|
|
||||||
song ID.
|
|
||||||
:raises ValueError: When an exported song has no lyrics, or
|
|
||||||
the per-ID extras name a song the working store does not
|
|
||||||
have.
|
|
||||||
"""
|
|
||||||
lines: list[str] = []
|
|
||||||
exported: set[str] = set()
|
|
||||||
song: Song
|
|
||||||
for song in session.scalars(sa.select(Song).order_by(Song.id)):
|
|
||||||
song_id: str = f"song-{song.id}"
|
|
||||||
if extras_per_id is not None and song_id not in extras_per_id:
|
|
||||||
continue
|
|
||||||
if song.lyrics is None:
|
|
||||||
raise ValueError(
|
|
||||||
f"song {song.id} \"{song.title}\": no lyrics")
|
|
||||||
song_extras: dict[str, Any] | None = None \
|
|
||||||
if extras_per_id is None else extras_per_id[song_id]
|
|
||||||
content: str = __build_content(
|
|
||||||
song.lyrics, extras, song_extras)
|
|
||||||
record: dict[str, str] = {
|
|
||||||
"id": song_id, "content": content}
|
|
||||||
lines.append(json.dumps(record, ensure_ascii=False))
|
|
||||||
exported.add(song_id)
|
|
||||||
if extras_per_id is not None:
|
|
||||||
missing: list[str] = sorted(set(extras_per_id) - exported)
|
|
||||||
if len(missing) > 0:
|
|
||||||
raise ValueError(
|
|
||||||
"the per-ID extras name songs the working store"
|
|
||||||
f" does not have: {', '.join(missing)}")
|
|
||||||
return lines
|
|
||||||
|
|
||||||
|
|
||||||
def main(argv: list[str] | None = None) -> int:
|
def main(argv: list[str] | None = None) -> int:
|
||||||
"""Export the LLM input JSONL file from the working store.
|
"""Export the LLM input JSONL file from the working store.
|
||||||
|
|
||||||
Every song is exported, unless ``--extras-per-id`` is given,
|
|
||||||
in which case only the songs its file names are.
|
|
||||||
|
|
||||||
:param argv: The command-line arguments, or None for
|
:param argv: The command-line arguments, or None for
|
||||||
``sys.argv``.
|
``sys.argv``.
|
||||||
:return: The exit status: 0 on success, non-zero on failure.
|
:return: The exit status: 0 on success, non-zero on failure.
|
||||||
"""
|
"""
|
||||||
started: float = time.monotonic()
|
started: float = time.monotonic()
|
||||||
args: argparse.Namespace = parse_args(argv)
|
args: argparse.Namespace = parse_args(argv)
|
||||||
session: Session = ds.get_db()
|
|
||||||
lines: list[str]
|
|
||||||
try:
|
try:
|
||||||
extras: dict[str, Any] | None = None
|
count: int = LlmInputExporter(
|
||||||
if args.extras is not None:
|
args.output_jsonl, args.extras,
|
||||||
extras = load_extras(args.extras)
|
args.extras_per_id).run()
|
||||||
extras_per_id: dict[str, dict[str, Any]] | None = None
|
|
||||||
if args.extras_per_id is not None:
|
|
||||||
extras_per_id = load_extras_per_id(args.extras_per_id)
|
|
||||||
lines = build_lines(session, extras, extras_per_id)
|
|
||||||
except (OSError, sa.exc.SQLAlchemyError, ValueError) as error:
|
except (OSError, sa.exc.SQLAlchemyError, ValueError) as error:
|
||||||
print(f"error: {error}", file=sys.stderr)
|
print(f"error: {error}", file=sys.stderr)
|
||||||
return 1
|
return 1
|
||||||
finally:
|
|
||||||
session.close()
|
|
||||||
args.output_jsonl.parent.mkdir(parents=True, exist_ok=True)
|
|
||||||
with open(args.output_jsonl, "w", encoding="utf-8") as file:
|
|
||||||
line: str
|
|
||||||
for line in lines:
|
|
||||||
file.write(line + "\n")
|
|
||||||
elapsed: str = format_duration(time.monotonic() - started)
|
elapsed: str = format_duration(time.monotonic() - started)
|
||||||
print(f"Done. {len(lines)} songs exported."
|
print(f"Done. {count} songs exported."
|
||||||
f" {elapsed} elapsed.", file=sys.stderr)
|
f" {elapsed} elapsed.", file=sys.stderr)
|
||||||
return 0
|
return 0
|
||||||
|
|||||||
@@ -12,12 +12,10 @@ layer. The working store is only read, never written; the
|
|||||||
``build-db`` subcommand assembles the captured files into the
|
``build-db`` subcommand assembles the captured files into the
|
||||||
store on the next rebuild.
|
store on the next rebuild.
|
||||||
|
|
||||||
Every fetched row is meant for later human verification: the
|
An unresolved artist or an error on one artist is noted on its
|
||||||
description of the resolved item is recorded in the note column
|
row and does not fail the run. A row whose name is no longer an
|
||||||
so that a bad match can be spotted. An unresolved artist or an
|
artist of the store is dropped from the snapshot and reported on
|
||||||
error on one artist is noted on its row and does not fail the
|
the standard error.
|
||||||
run. A row whose name is no longer an artist of the store is
|
|
||||||
dropped from the snapshot and reported on the standard error.
|
|
||||||
"""
|
"""
|
||||||
import argparse
|
import argparse
|
||||||
import csv
|
import csv
|
||||||
@@ -33,7 +31,7 @@ import urllib.request
|
|||||||
from collections.abc import Container, Sequence
|
from collections.abc import Container, Sequence
|
||||||
from dataclasses import asdict, dataclass, field, fields
|
from dataclasses import asdict, dataclass, field, fields
|
||||||
from pathlib import Path
|
from pathlib import Path
|
||||||
from typing import Any, Literal, TextIO
|
from typing import Any, ClassVar, Literal, TextIO
|
||||||
|
|
||||||
import sqlalchemy as sa
|
import sqlalchemy as sa
|
||||||
from sqlalchemy.orm import Session
|
from sqlalchemy.orm import Session
|
||||||
@@ -43,71 +41,6 @@ from ..database import ds
|
|||||||
from ..models import Artist, Song, SongArtist
|
from ..models import Artist, Song, SongArtist
|
||||||
from ..utils import format_duration
|
from ..utils import format_duration
|
||||||
|
|
||||||
API_URL: str = "https://www.wikidata.org/w/api.php"
|
|
||||||
"""The URL of the Wikidata API endpoint."""
|
|
||||||
SPARQL_URL: str = "https://query.wikidata.org/sparql"
|
|
||||||
"""The URL of the Wikidata Query Service SPARQL endpoint."""
|
|
||||||
USER_AGENT: str = (
|
|
||||||
f"pop-fem-audit-tools/{VERSION}"
|
|
||||||
" (https://github.com/imacat/pop-fem-audit;"
|
|
||||||
" mailto:imacat@mail.imacat.idv.tw)")
|
|
||||||
"""The User-Agent header sent on every HTTP request."""
|
|
||||||
TIMEOUT: float = 30.0
|
|
||||||
"""The timeout of an API HTTP request, in seconds."""
|
|
||||||
SPARQL_TIMEOUT: float = 90.0
|
|
||||||
"""The timeout of a SPARQL HTTP request, in seconds.
|
|
||||||
|
|
||||||
Higher than the API timeout: the WDQS server aborts a slow
|
|
||||||
query at 60 seconds, and a lower client timeout would race
|
|
||||||
that server-side abort and misclassify a slow-but-answerable
|
|
||||||
query as a client-side timeout instead of letting the server's
|
|
||||||
own HTTP error response arrive and enter the retry path."""
|
|
||||||
SLEEP_SECONDS: float = 1.0
|
|
||||||
"""The delay between consecutive HTTP requests, in seconds."""
|
|
||||||
MAX_ATTEMPTS: int = 5
|
|
||||||
"""The maximum number of attempts on a transient error."""
|
|
||||||
RETRY_SECONDS: float = 15.0
|
|
||||||
"""The back-off unit on a transient error, in seconds;
|
|
||||||
multiplied by the attempt number already made."""
|
|
||||||
RETRY_STATUSES: frozenset[int] = frozenset({429, 500, 502, 503})
|
|
||||||
"""The HTTP statuses that are retried with a back-off."""
|
|
||||||
MAX_STAGE1_TITLES: int = 3
|
|
||||||
"""The maximum number of charted titles used for the stage-1 song
|
|
||||||
corroboration."""
|
|
||||||
HUMAN_QID: str = "Q5"
|
|
||||||
"""The Wikidata item ID of "human"."""
|
|
||||||
ENSEMBLE_QID: str = "Q2088357"
|
|
||||||
"""The Wikidata item ID of "musical ensemble"."""
|
|
||||||
ORIGINAL_CAST_QID: str = "Q106497009"
|
|
||||||
"""The Wikidata item ID of "original cast"."""
|
|
||||||
GROUP_KEYWORDS: Sequence[str] = ("band", "group", "duo", "trio")
|
|
||||||
"""The label keywords that suggest a musical ensemble, covering
|
|
||||||
labels like "boy band" and "girl group"."""
|
|
||||||
NOTE_NOT_FOUND: str = "not found"
|
|
||||||
"""The note sentinel of an artist without a resolved Wikidata
|
|
||||||
item, written to the snapshot and read back for the
|
|
||||||
classification."""
|
|
||||||
CORPUS_START_YEAR: int = 2016
|
|
||||||
"""The first year of the corpus window: a member who left a group
|
|
||||||
before it never performed a corpus song."""
|
|
||||||
MIXED_GENDER: str = "mixed"
|
|
||||||
"""The gender recorded for a group whose members do not share one
|
|
||||||
gender."""
|
|
||||||
TIME_YEAR_PATTERN: re.Pattern[str] = re.compile(r"^[+-]?\d+")
|
|
||||||
"""The leading year of a Wikidata time value."""
|
|
||||||
PINNED_QIDS: dict[str, str] = {}
|
|
||||||
"""The last-resort pinned item IDs, keyed by the artist name.
|
|
||||||
|
|
||||||
An entry is for an artist the algorithm documented on
|
|
||||||
``ArtistFetcher`` is structurally unable to resolve, with its
|
|
||||||
justification recorded here. Currently empty: the only pin ever
|
|
||||||
needed, "Pinkfong" (typed as a brand, which the type gate
|
|
||||||
excludes by design), became moot when the store's artist entity
|
|
||||||
behind that credit was identified as Hope Segoine.
|
|
||||||
|
|
||||||
A pinned name skips the candidate retrieval and corroboration
|
|
||||||
steps; its item ID is used directly."""
|
|
||||||
|
|
||||||
|
|
||||||
class ArtistType(enum.StrEnum):
|
class ArtistType(enum.StrEnum):
|
||||||
"""The decided artist type of a snapshot row."""
|
"""The decided artist type of a snapshot row."""
|
||||||
@@ -147,15 +80,14 @@ class ArtistSnapshot:
|
|||||||
return asdict(self)
|
return asdict(self)
|
||||||
|
|
||||||
|
|
||||||
SNAPSHOT_FIELDS: Sequence[str] = tuple(
|
|
||||||
x.name for x in fields(ArtistSnapshot))
|
|
||||||
"""The header columns of the Wikidata artist snapshot CSV file."""
|
|
||||||
|
|
||||||
|
|
||||||
@dataclass
|
@dataclass
|
||||||
class GroupMember:
|
class GroupMember:
|
||||||
"""One has-part member of a Wikidata group item."""
|
"""One has-part member of a Wikidata group item."""
|
||||||
|
|
||||||
|
__CORPUS_START_YEAR: ClassVar[int] = 2016
|
||||||
|
"""The first year of the corpus window: a member who left a
|
||||||
|
group before it never performed a corpus song."""
|
||||||
|
|
||||||
qid: str
|
qid: str
|
||||||
"""The item ID of the member."""
|
"""The item ID of the member."""
|
||||||
start_years: list[int] = field(default_factory=list)
|
start_years: list[int] = field(default_factory=list)
|
||||||
@@ -181,7 +113,7 @@ class GroupMember:
|
|||||||
if len(self.end_years) == 0:
|
if len(self.end_years) == 0:
|
||||||
return True
|
return True
|
||||||
last_end: int = max(self.end_years)
|
last_end: int = max(self.end_years)
|
||||||
if last_end >= CORPUS_START_YEAR:
|
if last_end >= self.__CORPUS_START_YEAR:
|
||||||
return True
|
return True
|
||||||
if len(self.start_years) == 0:
|
if len(self.start_years) == 0:
|
||||||
return False
|
return False
|
||||||
@@ -229,20 +161,8 @@ class RetryExhausted(Exception):
|
|||||||
"""
|
"""
|
||||||
|
|
||||||
|
|
||||||
def parse_args(argv: list[str] | None) -> argparse.Namespace:
|
NOTE_NOT_FOUND: str = "not found"
|
||||||
"""Parse the command-line arguments.
|
"""The note marking an artist that could not be resolved."""
|
||||||
|
|
||||||
:param argv: The command-line arguments, or None for
|
|
||||||
``sys.argv``.
|
|
||||||
:return: The parsed arguments.
|
|
||||||
"""
|
|
||||||
parser: argparse.ArgumentParser = argparse.ArgumentParser(
|
|
||||||
description="Fetch the artist metadata from Wikidata"
|
|
||||||
" into the capture layer.")
|
|
||||||
parser.add_argument(
|
|
||||||
"wikidata_csv", type=Path,
|
|
||||||
help="the Wikidata artist snapshot CSV file")
|
|
||||||
return parser.parse_args(argv)
|
|
||||||
|
|
||||||
|
|
||||||
class ArtistFetcher:
|
class ArtistFetcher:
|
||||||
@@ -291,6 +211,71 @@ class ArtistFetcher:
|
|||||||
in the note.
|
in the note.
|
||||||
"""
|
"""
|
||||||
|
|
||||||
|
__API_URL: ClassVar[str] = "https://www.wikidata.org/w/api.php"
|
||||||
|
"""The URL of the Wikidata API endpoint."""
|
||||||
|
__SPARQL_URL: ClassVar[str] \
|
||||||
|
= "https://query.wikidata.org/sparql"
|
||||||
|
"""The URL of the Wikidata Query Service SPARQL endpoint."""
|
||||||
|
__USER_AGENT: ClassVar[str] = (
|
||||||
|
f"pop-fem-audit-tools/{VERSION}"
|
||||||
|
" (https://github.com/imacat/pop-fem-audit;"
|
||||||
|
" mailto:imacat@mail.imacat.idv.tw)")
|
||||||
|
"""The User-Agent header sent on every HTTP request."""
|
||||||
|
__TIMEOUT: ClassVar[float] = 30.0
|
||||||
|
"""The timeout of an API HTTP request, in seconds."""
|
||||||
|
__SPARQL_TIMEOUT: ClassVar[float] = 90.0
|
||||||
|
"""The timeout of a SPARQL HTTP request, in seconds.
|
||||||
|
|
||||||
|
Higher than the API timeout: the WDQS server aborts a slow
|
||||||
|
query at 60 seconds, and a lower client timeout would race
|
||||||
|
that server-side abort and misclassify a slow-but-answerable
|
||||||
|
query as a client-side timeout instead of letting the
|
||||||
|
server's own HTTP error response arrive and enter the retry
|
||||||
|
path."""
|
||||||
|
__SLEEP_SECONDS: ClassVar[float] = 1.0
|
||||||
|
"""The delay between consecutive HTTP requests, in
|
||||||
|
seconds."""
|
||||||
|
__MAX_ATTEMPTS: ClassVar[int] = 5
|
||||||
|
"""The maximum number of attempts on a transient error."""
|
||||||
|
__RETRY_SECONDS: ClassVar[float] = 15.0
|
||||||
|
"""The back-off unit on a transient error, in seconds;
|
||||||
|
multiplied by the attempt number already made."""
|
||||||
|
__RETRY_STATUSES: ClassVar[frozenset[int]] \
|
||||||
|
= frozenset({429, 500, 502, 503})
|
||||||
|
"""The HTTP statuses that are retried with a back-off."""
|
||||||
|
__MAX_STAGE1_TITLES: ClassVar[int] = 3
|
||||||
|
"""The maximum number of charted titles used for the
|
||||||
|
stage-1 song corroboration."""
|
||||||
|
__HUMAN_QID: ClassVar[str] = "Q5"
|
||||||
|
"""The Wikidata item ID of "human"."""
|
||||||
|
__ENSEMBLE_QID: ClassVar[str] = "Q2088357"
|
||||||
|
"""The Wikidata item ID of "musical ensemble"."""
|
||||||
|
__ORIGINAL_CAST_QID: ClassVar[str] = "Q106497009"
|
||||||
|
"""The Wikidata item ID of "original cast"."""
|
||||||
|
__GROUP_KEYWORDS: ClassVar[Sequence[str]] \
|
||||||
|
= ("band", "group", "duo", "trio")
|
||||||
|
"""The label keywords that suggest a musical ensemble,
|
||||||
|
covering labels like "boy band" and "girl group"."""
|
||||||
|
__MIXED_GENDER: ClassVar[str] = "mixed"
|
||||||
|
"""The gender recorded for a group whose members do not
|
||||||
|
share one gender."""
|
||||||
|
__TIME_YEAR_PATTERN: ClassVar[re.Pattern[str]] \
|
||||||
|
= re.compile(r"^[+-]?\d+")
|
||||||
|
"""The leading year of a Wikidata time value."""
|
||||||
|
__PINNED_QIDS: ClassVar[dict[str, str]] = {}
|
||||||
|
"""The last-resort pinned item IDs, keyed by the artist name.
|
||||||
|
|
||||||
|
An entry is for an artist the algorithm documented on
|
||||||
|
``ArtistFetcher`` is structurally unable to resolve, with its
|
||||||
|
justification recorded here. Currently empty: the only pin
|
||||||
|
ever needed, "Pinkfong" (typed as a brand, which the type
|
||||||
|
gate excludes by design), became moot when the store's
|
||||||
|
artist entity behind that credit was identified as Hope
|
||||||
|
Segoine.
|
||||||
|
|
||||||
|
A pinned name skips the candidate retrieval and corroboration
|
||||||
|
steps; its item ID is used directly."""
|
||||||
|
|
||||||
def __init__(self) -> None:
|
def __init__(self) -> None:
|
||||||
"""Construct the fetcher."""
|
"""Construct the fetcher."""
|
||||||
self.__sent: int = 0
|
self.__sent: int = 0
|
||||||
@@ -342,8 +327,8 @@ class ArtistFetcher:
|
|||||||
transient error are exhausted.
|
transient error are exhausted.
|
||||||
:raises ValueError: On a JSON decoding error.
|
:raises ValueError: On a JSON decoding error.
|
||||||
"""
|
"""
|
||||||
if name in PINNED_QIDS:
|
if name in self.__PINNED_QIDS:
|
||||||
return PINNED_QIDS[name]
|
return self.__PINNED_QIDS[name]
|
||||||
candidates: list[str] = self.__candidates(name)
|
candidates: list[str] = self.__candidates(name)
|
||||||
if len(candidates) == 0:
|
if len(candidates) == 0:
|
||||||
return None
|
return None
|
||||||
@@ -373,11 +358,11 @@ class ArtistFetcher:
|
|||||||
{{ ?item rdfs:label ?name }}
|
{{ ?item rdfs:label ?name }}
|
||||||
UNION {{ ?item skos:altLabel ?name }}
|
UNION {{ ?item skos:altLabel ?name }}
|
||||||
{{
|
{{
|
||||||
?item wdt:P31 wd:{HUMAN_QID}
|
?item wdt:P31 wd:{self.__HUMAN_QID}
|
||||||
}} UNION {{
|
}} UNION {{
|
||||||
?item wdt:P31/wdt:P279* wd:{ENSEMBLE_QID}
|
?item wdt:P31/wdt:P279* wd:{self.__ENSEMBLE_QID}
|
||||||
}} UNION {{
|
}} UNION {{
|
||||||
?item wdt:P31 wd:{ORIGINAL_CAST_QID}
|
?item wdt:P31 wd:{self.__ORIGINAL_CAST_QID}
|
||||||
}}
|
}}
|
||||||
}}
|
}}
|
||||||
"""
|
"""
|
||||||
@@ -403,7 +388,7 @@ class ArtistFetcher:
|
|||||||
transient error are exhausted.
|
transient error are exhausted.
|
||||||
:raises ValueError: On a JSON decoding error.
|
:raises ValueError: On a JSON decoding error.
|
||||||
"""
|
"""
|
||||||
subset: Sequence[str] = titles[:MAX_STAGE1_TITLES]
|
subset: Sequence[str] = titles[:self.__MAX_STAGE1_TITLES]
|
||||||
if len(subset) == 0:
|
if len(subset) == 0:
|
||||||
return None
|
return None
|
||||||
query: str = f"""
|
query: str = f"""
|
||||||
@@ -532,7 +517,7 @@ class ArtistFetcher:
|
|||||||
qid: str
|
qid: str
|
||||||
for qid in qids:
|
for qid in qids:
|
||||||
member: MemberClaims = claims.get(qid, MemberClaims())
|
member: MemberClaims = claims.get(qid, MemberClaims())
|
||||||
if HUMAN_QID not in member.instance_of_ids:
|
if self.__HUMAN_QID not in member.instance_of_ids:
|
||||||
continue
|
continue
|
||||||
if len(member.gender_ids) == 0:
|
if len(member.gender_ids) == 0:
|
||||||
return
|
return
|
||||||
@@ -543,7 +528,7 @@ class ArtistFetcher:
|
|||||||
[x[1] for x in genders], any_language=True)
|
[x[1] for x in genders], any_language=True)
|
||||||
unique: set[str] = {x[1] for x in genders}
|
unique: set[str] = {x[1] for x in genders}
|
||||||
snapshot.gender = labels[genders[0][1]] \
|
snapshot.gender = labels[genders[0][1]] \
|
||||||
if len(unique) == 1 else MIXED_GENDER
|
if len(unique) == 1 else self.__MIXED_GENDER
|
||||||
basis: str = "gender derived from members: " + "; ".join(
|
basis: str = "gender derived from members: " + "; ".join(
|
||||||
f"{x} {labels[y]}" for x, y in genders)
|
f"{x} {labels[y]}" for x, y in genders)
|
||||||
snapshot.note = f"{snapshot.note}; {basis}" \
|
snapshot.note = f"{snapshot.note}; {basis}" \
|
||||||
@@ -677,7 +662,8 @@ class ArtistFetcher:
|
|||||||
or not isinstance(value.get("time"), str):
|
or not isinstance(value.get("time"), str):
|
||||||
continue
|
continue
|
||||||
match: re.Match[str] | None \
|
match: re.Match[str] | None \
|
||||||
= TIME_YEAR_PATTERN.match(value["time"])
|
= ArtistFetcher.__TIME_YEAR_PATTERN.match(
|
||||||
|
value["time"])
|
||||||
if match is not None:
|
if match is not None:
|
||||||
years.append(int(match.group()))
|
years.append(int(match.group()))
|
||||||
return years
|
return years
|
||||||
@@ -806,12 +792,13 @@ class ArtistFetcher:
|
|||||||
``ArtistType.GROUP`` for a musical ensemble, or the
|
``ArtistType.GROUP`` for a musical ensemble, or the
|
||||||
empty string for the human to decide.
|
empty string for the human to decide.
|
||||||
"""
|
"""
|
||||||
if HUMAN_QID in type_ids:
|
if ArtistFetcher.__HUMAN_QID in type_ids:
|
||||||
return ArtistType.SOLO
|
return ArtistType.SOLO
|
||||||
qid: str
|
qid: str
|
||||||
for qid in type_ids:
|
for qid in type_ids:
|
||||||
label: str = labels.get(qid, "").lower()
|
label: str = labels.get(qid, "").lower()
|
||||||
if any(x in label for x in GROUP_KEYWORDS):
|
if any(x in label
|
||||||
|
for x in ArtistFetcher.__GROUP_KEYWORDS):
|
||||||
return ArtistType.GROUP
|
return ArtistType.GROUP
|
||||||
return ""
|
return ""
|
||||||
|
|
||||||
@@ -827,14 +814,15 @@ class ArtistFetcher:
|
|||||||
transient error are exhausted.
|
transient error are exhausted.
|
||||||
:raises ValueError: On a JSON decoding error.
|
:raises ValueError: On a JSON decoding error.
|
||||||
"""
|
"""
|
||||||
url: str = (f"{SPARQL_URL}?"
|
url: str = (
|
||||||
f"{urllib.parse.urlencode({'query': query})}")
|
f"{self.__SPARQL_URL}?"
|
||||||
|
f"{urllib.parse.urlencode({'query': query})}")
|
||||||
request: urllib.request.Request = urllib.request.Request(
|
request: urllib.request.Request = urllib.request.Request(
|
||||||
url, headers={
|
url, headers={
|
||||||
"User-Agent": USER_AGENT,
|
"User-Agent": self.__USER_AGENT,
|
||||||
"Accept": "application/sparql-results+json"})
|
"Accept": "application/sparql-results+json"})
|
||||||
body: bytes = self.__send(
|
body: bytes = self.__send(
|
||||||
request, timeout=SPARQL_TIMEOUT)
|
request, timeout=self.__SPARQL_TIMEOUT)
|
||||||
data: Any = json.loads(body)
|
data: Any = json.loads(body)
|
||||||
bindings: Any = None
|
bindings: Any = None
|
||||||
if isinstance(data, dict) \
|
if isinstance(data, dict) \
|
||||||
@@ -868,13 +856,14 @@ class ArtistFetcher:
|
|||||||
transient error are exhausted.
|
transient error are exhausted.
|
||||||
:raises ValueError: On a JSON decoding error.
|
:raises ValueError: On a JSON decoding error.
|
||||||
"""
|
"""
|
||||||
url: str = f"{API_URL}?{urllib.parse.urlencode(params)}"
|
url: str \
|
||||||
|
= f"{self.__API_URL}?{urllib.parse.urlencode(params)}"
|
||||||
request: urllib.request.Request = urllib.request.Request(
|
request: urllib.request.Request = urllib.request.Request(
|
||||||
url, headers={"User-Agent": USER_AGENT})
|
url, headers={"User-Agent": self.__USER_AGENT})
|
||||||
return json.loads(self.__send(request))
|
return json.loads(self.__send(request))
|
||||||
|
|
||||||
def __send(self, request: urllib.request.Request,
|
def __send(self, request: urllib.request.Request,
|
||||||
timeout: float = TIMEOUT) -> bytes:
|
timeout: float = __TIMEOUT) -> bytes:
|
||||||
"""Send an HTTP request, retrying on a transient error.
|
"""Send an HTTP request, retrying on a transient error.
|
||||||
|
|
||||||
Consecutive requests are separated by a fixed delay. A
|
Consecutive requests are separated by a fixed delay. A
|
||||||
@@ -892,7 +881,7 @@ class ArtistFetcher:
|
|||||||
transient error are exhausted.
|
transient error are exhausted.
|
||||||
"""
|
"""
|
||||||
if self.__sent > 0:
|
if self.__sent > 0:
|
||||||
time.sleep(SLEEP_SECONDS)
|
time.sleep(self.__SLEEP_SECONDS)
|
||||||
self.__sent += 1
|
self.__sent += 1
|
||||||
attempt: int = 1
|
attempt: int = 1
|
||||||
reason: str | None
|
reason: str | None
|
||||||
@@ -905,10 +894,10 @@ class ArtistFetcher:
|
|||||||
reason = self.__retry_reason(error)
|
reason = self.__retry_reason(error)
|
||||||
if reason is None:
|
if reason is None:
|
||||||
raise
|
raise
|
||||||
if attempt >= MAX_ATTEMPTS:
|
if attempt >= self.__MAX_ATTEMPTS:
|
||||||
raise RetryExhausted(
|
raise RetryExhausted(
|
||||||
f"retries exhausted ({reason})") from error
|
f"retries exhausted ({reason})") from error
|
||||||
time.sleep(RETRY_SECONDS * attempt)
|
time.sleep(self.__RETRY_SECONDS * attempt)
|
||||||
attempt += 1
|
attempt += 1
|
||||||
|
|
||||||
@staticmethod
|
@staticmethod
|
||||||
@@ -923,7 +912,7 @@ class ArtistFetcher:
|
|||||||
the error is not transient and must not be retried.
|
the error is not transient and must not be retried.
|
||||||
"""
|
"""
|
||||||
if isinstance(error, urllib.error.HTTPError):
|
if isinstance(error, urllib.error.HTTPError):
|
||||||
if error.code not in RETRY_STATUSES:
|
if error.code not in ArtistFetcher.__RETRY_STATUSES:
|
||||||
return None
|
return None
|
||||||
return str(error)
|
return str(error)
|
||||||
if isinstance(error, TimeoutError):
|
if isinstance(error, TimeoutError):
|
||||||
@@ -968,92 +957,216 @@ class ArtistFetcher:
|
|||||||
return uri.rsplit("/", 1)[-1]
|
return uri.rsplit("/", 1)[-1]
|
||||||
|
|
||||||
|
|
||||||
def read_snapshot_rows(file: TextIO) -> list[dict[str, str]]:
|
@dataclass(frozen=True)
|
||||||
"""Read the current rows of a snapshot CSV file handle.
|
class FetchCounts:
|
||||||
|
"""The outcome counts of one snapshot update run."""
|
||||||
|
|
||||||
:param file: The open, seekable snapshot CSV file.
|
fetched: int
|
||||||
:return: The rows, keyed by the column name.
|
"""The number of artists newly resolved."""
|
||||||
:raises OSError: When the file cannot be read.
|
not_found: int
|
||||||
|
"""The number of artists left unresolved."""
|
||||||
|
errors: int
|
||||||
|
"""The number of artists that ended in an error."""
|
||||||
|
|
||||||
|
|
||||||
|
class ArtistSnapshotUpdater:
|
||||||
|
"""The updater of the Wikidata artist snapshot CSV file.
|
||||||
|
|
||||||
|
Fetches the metadata of every artist of the working store
|
||||||
|
that the snapshot does not resolve yet, appends a row for
|
||||||
|
each to the snapshot as it is fetched, and rewrites the
|
||||||
|
snapshot sorted by artist name with its stale rows dropped.
|
||||||
"""
|
"""
|
||||||
file.seek(0)
|
|
||||||
reader: csv.DictReader[str] = csv.DictReader(file)
|
|
||||||
return list(reader)
|
|
||||||
|
|
||||||
|
__SNAPSHOT_FIELDS: ClassVar[Sequence[str]] = tuple(
|
||||||
|
x.name for x in fields(ArtistSnapshot))
|
||||||
|
"""The header columns of the Wikidata artist snapshot CSV file."""
|
||||||
|
|
||||||
def read_artist_titles(session: Session,
|
def __init__(self, wikidata_csv: Path) -> None:
|
||||||
artist_id: int) -> list[str]:
|
"""Set up the updater.
|
||||||
"""Read the charted song titles credited to an artist.
|
|
||||||
|
|
||||||
:param session: The database session.
|
:param wikidata_csv: The Wikidata artist snapshot CSV
|
||||||
:param artist_id: The artist ID.
|
file.
|
||||||
:return: The song titles credited to the artist, ordered by
|
"""
|
||||||
the song ID, with the duplicate titles removed.
|
self.__wikidata_csv: Path = wikidata_csv
|
||||||
"""
|
"""The Wikidata artist snapshot CSV file."""
|
||||||
titles: Sequence[str] = session.scalars(
|
|
||||||
sa.select(Song.title)
|
|
||||||
.join(SongArtist, SongArtist.song_id == Song.id)
|
|
||||||
.where(SongArtist.artist_id == artist_id)
|
|
||||||
.order_by(Song.id)).all()
|
|
||||||
return list(dict.fromkeys(titles))
|
|
||||||
|
|
||||||
|
def run(self) -> FetchCounts:
|
||||||
|
"""Fetch every unresolved artist and update the snapshot.
|
||||||
|
|
||||||
def ensure_snapshot_header(file: TextIO) -> None:
|
:return: The counts of the run.
|
||||||
"""Write the snapshot CSV header row if the file is empty.
|
:raises OSError: When the snapshot file, or its parent
|
||||||
|
directory, cannot be read or written.
|
||||||
|
:raises sqlalchemy.exc.SQLAlchemyError: When the working
|
||||||
|
store cannot be read.
|
||||||
|
"""
|
||||||
|
session: Session = ds.get_db()
|
||||||
|
try:
|
||||||
|
return self.__run(session)
|
||||||
|
finally:
|
||||||
|
session.close()
|
||||||
|
|
||||||
:param file: The open, seekable snapshot CSV file.
|
def __run(self, session: Session) -> FetchCounts:
|
||||||
:return: None.
|
"""Run the fetch loop with an open database session.
|
||||||
:raises OSError: When the file cannot be written.
|
|
||||||
"""
|
:param session: The database session.
|
||||||
file.seek(0, os.SEEK_END)
|
:return: The counts of the run.
|
||||||
if file.tell() == 0:
|
:raises OSError: When the snapshot file, or its parent
|
||||||
csv.writer(file).writerow(SNAPSHOT_FIELDS)
|
directory, cannot be read or written.
|
||||||
|
"""
|
||||||
|
fetcher: ArtistFetcher = ArtistFetcher()
|
||||||
|
fetched: int = 0
|
||||||
|
not_found: int = 0
|
||||||
|
errors: int = 0
|
||||||
|
self.__wikidata_csv.parent.mkdir(
|
||||||
|
parents=True, exist_ok=True)
|
||||||
|
with open(self.__wikidata_csv, "a+", encoding="utf-8",
|
||||||
|
newline="") as csv_file:
|
||||||
|
done: set[str] = {
|
||||||
|
x["name"] for x in
|
||||||
|
self.__read_snapshot_rows(csv_file)
|
||||||
|
if x["gender"] != ""}
|
||||||
|
self.__ensure_snapshot_header(csv_file)
|
||||||
|
names: set[str] = set()
|
||||||
|
artist: Artist
|
||||||
|
for artist in session.scalars(
|
||||||
|
sa.select(Artist).order_by(Artist.id)):
|
||||||
|
names.add(artist.name)
|
||||||
|
if artist.name in done:
|
||||||
|
continue
|
||||||
|
titles: list[str] = self.__read_artist_titles(
|
||||||
|
session, artist.id)
|
||||||
|
snapshot: ArtistSnapshot = fetcher.fetch(
|
||||||
|
artist.name, titles)
|
||||||
|
self.__append_row(csv_file, snapshot)
|
||||||
|
status: str = snapshot.qid
|
||||||
|
if snapshot.note == NOTE_NOT_FOUND:
|
||||||
|
not_found += 1
|
||||||
|
status = NOTE_NOT_FOUND
|
||||||
|
elif snapshot.note.startswith("error: "):
|
||||||
|
errors += 1
|
||||||
|
status = snapshot.note
|
||||||
|
else:
|
||||||
|
fetched += 1
|
||||||
|
print(f"artist \"{artist.name}\": {status}",
|
||||||
|
file=sys.stderr)
|
||||||
|
self.__write_snapshot(csv_file, names)
|
||||||
|
return FetchCounts(
|
||||||
|
fetched=fetched, not_found=not_found, errors=errors)
|
||||||
|
|
||||||
|
@staticmethod
|
||||||
|
def __read_snapshot_rows(file: TextIO) \
|
||||||
|
-> list[dict[str, str]]:
|
||||||
|
"""Read the current rows of a snapshot CSV file handle.
|
||||||
|
|
||||||
|
:param file: The open, seekable snapshot CSV file.
|
||||||
|
:return: The rows, keyed by the column name.
|
||||||
|
:raises OSError: When the file cannot be read.
|
||||||
|
"""
|
||||||
|
file.seek(0)
|
||||||
|
reader: csv.DictReader[str] = csv.DictReader(file)
|
||||||
|
return list(reader)
|
||||||
|
|
||||||
|
@staticmethod
|
||||||
|
def __read_artist_titles(session: Session,
|
||||||
|
artist_id: int) -> list[str]:
|
||||||
|
"""Read the charted song titles credited to an artist.
|
||||||
|
|
||||||
|
:param session: The database session.
|
||||||
|
:param artist_id: The artist ID.
|
||||||
|
:return: The song titles credited to the artist, ordered
|
||||||
|
by the song ID, with the duplicate titles removed.
|
||||||
|
"""
|
||||||
|
titles: Sequence[str] = session.scalars(
|
||||||
|
sa.select(Song.title)
|
||||||
|
.join(SongArtist, SongArtist.song_id == Song.id)
|
||||||
|
.where(SongArtist.artist_id == artist_id)
|
||||||
|
.order_by(Song.id)).all()
|
||||||
|
return list(dict.fromkeys(titles))
|
||||||
|
|
||||||
|
@staticmethod
|
||||||
|
def __ensure_snapshot_header(file: TextIO) -> None:
|
||||||
|
"""Write the snapshot CSV header row if the file is
|
||||||
|
empty.
|
||||||
|
|
||||||
|
:param file: The open, seekable snapshot CSV file.
|
||||||
|
:return: None.
|
||||||
|
:raises OSError: When the file cannot be written.
|
||||||
|
"""
|
||||||
|
file.seek(0, os.SEEK_END)
|
||||||
|
if file.tell() == 0:
|
||||||
|
csv.writer(file).writerow(
|
||||||
|
ArtistSnapshotUpdater.__SNAPSHOT_FIELDS)
|
||||||
|
file.flush()
|
||||||
|
|
||||||
|
@staticmethod
|
||||||
|
def __append_row(file: TextIO,
|
||||||
|
snapshot: ArtistSnapshot) -> None:
|
||||||
|
"""Append a snapshot row to a snapshot CSV file handle.
|
||||||
|
|
||||||
|
:param file: The open snapshot CSV file, opened for
|
||||||
|
append.
|
||||||
|
:param snapshot: The snapshot of an artist.
|
||||||
|
:return: None.
|
||||||
|
:raises OSError: When the file cannot be written.
|
||||||
|
"""
|
||||||
|
csv.DictWriter(
|
||||||
|
file, ArtistSnapshotUpdater.__SNAPSHOT_FIELDS).writerow(
|
||||||
|
snapshot.to_row())
|
||||||
file.flush()
|
file.flush()
|
||||||
|
|
||||||
|
@staticmethod
|
||||||
|
def __write_snapshot(file: TextIO, names: Container[str]) \
|
||||||
|
-> None:
|
||||||
|
"""Rewrite a snapshot CSV file handle sorted by artist
|
||||||
|
name.
|
||||||
|
|
||||||
def append_row(file: TextIO, snapshot: ArtistSnapshot) -> None:
|
The rows are ordered by the case-folded artist name,
|
||||||
"""Append a snapshot row to a snapshot CSV file handle.
|
matching the convention of the derived ``artists.csv``.
|
||||||
|
An artist keeps one row only, the last one of the file,
|
||||||
|
so that a re-fetched artist replaces its earlier row. A
|
||||||
|
row whose name is not an artist of the store is dropped
|
||||||
|
and reported on the standard error.
|
||||||
|
|
||||||
:param file: The open snapshot CSV file, opened for append.
|
:param file: The open, seekable snapshot CSV file.
|
||||||
:param snapshot: The snapshot of an artist.
|
:param names: The artist names of the working store.
|
||||||
:return: None.
|
:return: None.
|
||||||
:raises OSError: When the file cannot be written.
|
:raises OSError: When the file cannot be read or written.
|
||||||
|
"""
|
||||||
|
kept: dict[str, dict[str, str]] = {}
|
||||||
|
row: dict[str, str]
|
||||||
|
for row in ArtistSnapshotUpdater.__read_snapshot_rows(
|
||||||
|
file):
|
||||||
|
if row["name"] not in names:
|
||||||
|
print(f"dropped stale row \"{row['name']}\":"
|
||||||
|
" no such artist in the store",
|
||||||
|
file=sys.stderr)
|
||||||
|
continue
|
||||||
|
kept[row["name"]] = row
|
||||||
|
ordered: list[dict[str, str]] = sorted(
|
||||||
|
kept.values(), key=lambda x: x["name"].casefold())
|
||||||
|
file.seek(0)
|
||||||
|
file.truncate()
|
||||||
|
writer: csv.DictWriter[str] = csv.DictWriter(
|
||||||
|
file, ArtistSnapshotUpdater.__SNAPSHOT_FIELDS)
|
||||||
|
writer.writeheader()
|
||||||
|
writer.writerows(ordered)
|
||||||
|
|
||||||
|
|
||||||
|
def parse_args(argv: list[str] | None) -> argparse.Namespace:
|
||||||
|
"""Parse the command-line arguments.
|
||||||
|
|
||||||
|
:param argv: The command-line arguments, or None for
|
||||||
|
``sys.argv``.
|
||||||
|
:return: The parsed arguments.
|
||||||
"""
|
"""
|
||||||
csv.DictWriter(file, SNAPSHOT_FIELDS).writerow(
|
parser: argparse.ArgumentParser = argparse.ArgumentParser(
|
||||||
snapshot.to_row())
|
description="Fetch the artist metadata from Wikidata"
|
||||||
file.flush()
|
" into the capture layer.")
|
||||||
|
parser.add_argument(
|
||||||
|
"wikidata_csv", type=Path,
|
||||||
def write_snapshot(file: TextIO, names: Container[str]) -> None:
|
help="the Wikidata artist snapshot CSV file")
|
||||||
"""Rewrite a snapshot CSV file handle sorted by artist name.
|
return parser.parse_args(argv)
|
||||||
|
|
||||||
The rows are ordered by the case-folded artist name, matching
|
|
||||||
the convention of the derived ``artists.csv``. An artist
|
|
||||||
keeps one row only, the last one of the file, so that a
|
|
||||||
re-fetched artist replaces its earlier row. A row whose name
|
|
||||||
is not an artist of the store is dropped and reported on the
|
|
||||||
standard error.
|
|
||||||
|
|
||||||
:param file: The open, seekable snapshot CSV file.
|
|
||||||
:param names: The artist names of the working store.
|
|
||||||
:return: None.
|
|
||||||
:raises OSError: When the file cannot be read or written.
|
|
||||||
"""
|
|
||||||
kept: dict[str, dict[str, str]] = {}
|
|
||||||
row: dict[str, str]
|
|
||||||
for row in read_snapshot_rows(file):
|
|
||||||
if row["name"] not in names:
|
|
||||||
print(f"dropped stale row \"{row['name']}\":"
|
|
||||||
" no such artist in the store", file=sys.stderr)
|
|
||||||
continue
|
|
||||||
kept[row["name"]] = row
|
|
||||||
ordered: list[dict[str, str]] = sorted(
|
|
||||||
kept.values(), key=lambda x: x["name"].casefold())
|
|
||||||
file.seek(0)
|
|
||||||
file.truncate()
|
|
||||||
writer: csv.DictWriter[str] = csv.DictWriter(
|
|
||||||
file, SNAPSHOT_FIELDS)
|
|
||||||
writer.writeheader()
|
|
||||||
writer.writerows(ordered)
|
|
||||||
|
|
||||||
|
|
||||||
def main(argv: list[str] | None = None) -> int:
|
def main(argv: list[str] | None = None) -> int:
|
||||||
@@ -1066,52 +1179,16 @@ def main(argv: list[str] | None = None) -> int:
|
|||||||
"""
|
"""
|
||||||
started: float = time.monotonic()
|
started: float = time.monotonic()
|
||||||
args: argparse.Namespace = parse_args(argv)
|
args: argparse.Namespace = parse_args(argv)
|
||||||
fetcher: ArtistFetcher = ArtistFetcher()
|
|
||||||
fetched: int = 0
|
|
||||||
not_found: int = 0
|
|
||||||
errors: int = 0
|
|
||||||
session: Session = ds.get_db()
|
|
||||||
try:
|
try:
|
||||||
args.wikidata_csv.parent.mkdir(
|
counts: FetchCounts \
|
||||||
parents=True, exist_ok=True)
|
= ArtistSnapshotUpdater(args.wikidata_csv).run()
|
||||||
with open(args.wikidata_csv, "a+", encoding="utf-8",
|
|
||||||
newline="") as csv_file:
|
|
||||||
done: set[str] = {x["name"] for x in
|
|
||||||
read_snapshot_rows(csv_file)
|
|
||||||
if x["gender"] != ""}
|
|
||||||
ensure_snapshot_header(csv_file)
|
|
||||||
names: set[str] = set()
|
|
||||||
artist: Artist
|
|
||||||
for artist in session.scalars(
|
|
||||||
sa.select(Artist).order_by(Artist.id)):
|
|
||||||
names.add(artist.name)
|
|
||||||
if artist.name in done:
|
|
||||||
continue
|
|
||||||
titles: list[str] = read_artist_titles(
|
|
||||||
session, artist.id)
|
|
||||||
snapshot: ArtistSnapshot = fetcher.fetch(
|
|
||||||
artist.name, titles)
|
|
||||||
append_row(csv_file, snapshot)
|
|
||||||
status: str = snapshot.qid
|
|
||||||
if snapshot.note == NOTE_NOT_FOUND:
|
|
||||||
not_found += 1
|
|
||||||
status = "not found"
|
|
||||||
elif snapshot.note.startswith("error: "):
|
|
||||||
errors += 1
|
|
||||||
status = snapshot.note
|
|
||||||
else:
|
|
||||||
fetched += 1
|
|
||||||
print(f"artist \"{artist.name}\": {status}",
|
|
||||||
file=sys.stderr)
|
|
||||||
write_snapshot(csv_file, names)
|
|
||||||
except (OSError, sa.exc.SQLAlchemyError) as error:
|
except (OSError, sa.exc.SQLAlchemyError) as error:
|
||||||
print(f"error: {error}", file=sys.stderr)
|
print(f"error: {error}", file=sys.stderr)
|
||||||
return 1
|
return 1
|
||||||
finally:
|
attempted: int = counts.fetched + counts.not_found \
|
||||||
session.close()
|
+ counts.errors
|
||||||
attempted: int = fetched + not_found + errors
|
|
||||||
elapsed: str = format_duration(time.monotonic() - started)
|
elapsed: str = format_duration(time.monotonic() - started)
|
||||||
print(f"Done. Resolved {fetched}/{attempted} artists."
|
print(f"Done. Resolved {counts.fetched}/{attempted}"
|
||||||
f" {elapsed} elapsed.",
|
f" artists. {elapsed} elapsed.",
|
||||||
file=sys.stderr)
|
file=sys.stderr)
|
||||||
return 0
|
return 0
|
||||||
|
|||||||
@@ -30,8 +30,9 @@ import time
|
|||||||
import urllib.parse
|
import urllib.parse
|
||||||
import urllib.request
|
import urllib.request
|
||||||
from collections.abc import Sequence
|
from collections.abc import Sequence
|
||||||
|
from dataclasses import dataclass
|
||||||
from pathlib import Path
|
from pathlib import Path
|
||||||
from typing import Any
|
from typing import Any, ClassVar
|
||||||
|
|
||||||
import sqlalchemy as sa
|
import sqlalchemy as sa
|
||||||
from sqlalchemy.orm import Session
|
from sqlalchemy.orm import Session
|
||||||
@@ -45,60 +46,6 @@ from ..models import (
|
|||||||
)
|
)
|
||||||
from ..utils import format_duration
|
from ..utils import format_duration
|
||||||
|
|
||||||
PROVENANCE_FIELDS: Sequence[str] = (
|
|
||||||
"song_id", "source", "method", "acquired_at", "note")
|
|
||||||
"""The header columns of the lyrics provenance CSV file."""
|
|
||||||
USER_AGENT: str = ("pop-fem-audit-tools"
|
|
||||||
" (https://github.com/imacat/pop-fem-audit)")
|
|
||||||
"""The User-Agent header sent on every HTTP request."""
|
|
||||||
TIMEOUT: float = 30.0
|
|
||||||
"""The timeout of an HTTP request, in seconds."""
|
|
||||||
SLEEP_SECONDS: float = 1.0
|
|
||||||
"""The delay between consecutive HTTP requests, in seconds."""
|
|
||||||
|
|
||||||
|
|
||||||
def __build_normalization() -> dict[int, str | None]:
|
|
||||||
"""Build the lyrics normalization translation table.
|
|
||||||
|
|
||||||
:return: The codepoint-to-replacement mapping, a replacement
|
|
||||||
of None meaning removal.
|
|
||||||
"""
|
|
||||||
table: dict[int, str | None] = {}
|
|
||||||
codepoint: int
|
|
||||||
for codepoint in range(0x80, 0xa0):
|
|
||||||
try:
|
|
||||||
table[codepoint] = bytes([codepoint]).decode("cp1252")
|
|
||||||
except UnicodeDecodeError:
|
|
||||||
table[codepoint] = None
|
|
||||||
table[0x0435] = "e"
|
|
||||||
table[0x03cc] = "ó"
|
|
||||||
for codepoint in (0x2005, 0x205f, 0x200a):
|
|
||||||
table[codepoint] = " "
|
|
||||||
for codepoint in (0x200b, 0x200c, 0x200d, 0xfeff):
|
|
||||||
table[codepoint] = None
|
|
||||||
return table
|
|
||||||
|
|
||||||
|
|
||||||
NORMALIZATION: dict[int, str | None] = __build_normalization()
|
|
||||||
"""The codepoint-to-replacement mapping applied to fetched
|
|
||||||
lyrics: cp1252-mojibake restoration for U+0080-U+009F (with the
|
|
||||||
five byte values undefined in cp1252 removed), homoglyph
|
|
||||||
restoration for the Cyrillic "e" and the Greek "o" with tonos,
|
|
||||||
ASCII-space restoration for exotic space variants, and removal
|
|
||||||
of zero-width characters. A replacement of None removes the
|
|
||||||
codepoint."""
|
|
||||||
|
|
||||||
|
|
||||||
def normalize_lyrics(text: str) -> str:
|
|
||||||
"""Restore or remove watermark and mojibake characters.
|
|
||||||
|
|
||||||
:param text: The lyrics text as fetched from an API.
|
|
||||||
:return: The text with the codepoints in
|
|
||||||
:data:`NORMALIZATION` replaced or removed; every other
|
|
||||||
character is unchanged.
|
|
||||||
"""
|
|
||||||
return text.translate(NORMALIZATION)
|
|
||||||
|
|
||||||
|
|
||||||
def parse_args(argv: list[str] | None) -> argparse.Namespace:
|
def parse_args(argv: list[str] | None) -> argparse.Namespace:
|
||||||
"""Parse the command-line arguments.
|
"""Parse the command-line arguments.
|
||||||
@@ -122,6 +69,16 @@ def parse_args(argv: list[str] | None) -> argparse.Namespace:
|
|||||||
class LyricsFetcher:
|
class LyricsFetcher:
|
||||||
"""A fetcher of song lyrics from the public lyrics APIs."""
|
"""A fetcher of song lyrics from the public lyrics APIs."""
|
||||||
|
|
||||||
|
__USER_AGENT: ClassVar[str] = (
|
||||||
|
"pop-fem-audit-tools"
|
||||||
|
" (https://github.com/imacat/pop-fem-audit)")
|
||||||
|
"""The User-Agent header sent on every HTTP request."""
|
||||||
|
__TIMEOUT: ClassVar[float] = 30.0
|
||||||
|
"""The timeout of an HTTP request, in seconds."""
|
||||||
|
__SLEEP_SECONDS: ClassVar[float] = 1.0
|
||||||
|
"""The delay between consecutive HTTP requests, in
|
||||||
|
seconds."""
|
||||||
|
|
||||||
def __init__(self) -> None:
|
def __init__(self) -> None:
|
||||||
"""Construct the fetcher."""
|
"""Construct the fetcher."""
|
||||||
self.__sent: int = 0
|
self.__sent: int = 0
|
||||||
@@ -193,78 +150,216 @@ class LyricsFetcher:
|
|||||||
network, or decoding error.
|
network, or decoding error.
|
||||||
"""
|
"""
|
||||||
if self.__sent > 0:
|
if self.__sent > 0:
|
||||||
time.sleep(SLEEP_SECONDS)
|
time.sleep(self.__SLEEP_SECONDS)
|
||||||
self.__sent += 1
|
self.__sent += 1
|
||||||
request: urllib.request.Request = urllib.request.Request(
|
request: urllib.request.Request = urllib.request.Request(
|
||||||
url, headers={"User-Agent": USER_AGENT})
|
url, headers={"User-Agent": self.__USER_AGENT})
|
||||||
try:
|
try:
|
||||||
with urllib.request.urlopen(
|
with urllib.request.urlopen(
|
||||||
request, timeout=TIMEOUT) as response:
|
request, timeout=self.__TIMEOUT) as response:
|
||||||
return json.load(response)
|
return json.load(response)
|
||||||
except (OSError, ValueError):
|
except (OSError, ValueError):
|
||||||
return None
|
return None
|
||||||
|
|
||||||
|
|
||||||
def query_artist(session: Session, song_id: int) -> str:
|
@dataclass(frozen=True)
|
||||||
"""Find the artist name to query the APIs with.
|
class LyricsFetchCounts:
|
||||||
|
"""The outcome of one run of fetching the missing lyrics."""
|
||||||
|
|
||||||
:param session: The database session.
|
fetched: int
|
||||||
:param song_id: The song ID.
|
"""The number of songs newly fetched."""
|
||||||
:return: The name of the primary-role artist with the lowest
|
missed: int
|
||||||
position.
|
"""The number of songs every API missed."""
|
||||||
"""
|
|
||||||
name: str | None = session.scalar(
|
|
||||||
sa.select(Artist.name)
|
|
||||||
.join(SongArtist, SongArtist.artist_id == Artist.id)
|
|
||||||
.where(SongArtist.song_id == song_id,
|
|
||||||
SongArtist.role == Role.PRIMARY)
|
|
||||||
.order_by(SongArtist.position)
|
|
||||||
.limit(1))
|
|
||||||
assert name is not None
|
|
||||||
return name
|
|
||||||
|
|
||||||
|
|
||||||
def save_lyrics(lyrics_dir: Path, song_id: int,
|
class LyricsFetchRunner:
|
||||||
lyrics: str) -> None:
|
"""The orchestrator of one run of fetching missing lyrics."""
|
||||||
"""Write the lyrics of a song into the cache directory.
|
|
||||||
|
|
||||||
The cache directory is created when missing.
|
__PROVENANCE_FIELDS: ClassVar[Sequence[str]] = (
|
||||||
|
"song_id", "source", "method", "acquired_at", "note")
|
||||||
|
"""The header columns of the lyrics provenance CSV file."""
|
||||||
|
|
||||||
The lyrics text is normalized with :func:`normalize_lyrics`
|
@staticmethod
|
||||||
before being written.
|
def __build_normalization() -> dict[int, str | None]:
|
||||||
|
"""Build the lyrics normalization translation table.
|
||||||
|
|
||||||
:param lyrics_dir: The lyrics cache directory.
|
:return: The codepoint-to-replacement mapping, a
|
||||||
:param song_id: The song ID.
|
replacement of None meaning removal.
|
||||||
:param lyrics: The lyrics text.
|
"""
|
||||||
:return: None.
|
table: dict[int, str | None] = {}
|
||||||
:raises OSError: When the file cannot be written.
|
codepoint: int
|
||||||
"""
|
for codepoint in range(0x80, 0xa0):
|
||||||
lyrics_dir.mkdir(parents=True, exist_ok=True)
|
try:
|
||||||
(lyrics_dir / f"{song_id}.txt").write_text(
|
table[codepoint] = bytes(
|
||||||
normalize_lyrics(lyrics), encoding="utf-8")
|
[codepoint]).decode("cp1252")
|
||||||
|
except UnicodeDecodeError:
|
||||||
|
table[codepoint] = None
|
||||||
|
table[0x0435] = "e"
|
||||||
|
table[0x03cc] = "ó"
|
||||||
|
for codepoint in (0x2005, 0x205f, 0x200a):
|
||||||
|
table[codepoint] = " "
|
||||||
|
for codepoint in (0x200b, 0x200c, 0x200d, 0xfeff):
|
||||||
|
table[codepoint] = None
|
||||||
|
return table
|
||||||
|
|
||||||
|
__NORMALIZATION: ClassVar[dict[int, str | None]] \
|
||||||
|
= __build_normalization()
|
||||||
|
"""The codepoint-to-replacement mapping applied to fetched
|
||||||
|
lyrics: cp1252-mojibake restoration for U+0080-U+009F (with
|
||||||
|
the five byte values undefined in cp1252 removed), homoglyph
|
||||||
|
restoration for the Cyrillic "e" and the Greek "o" with
|
||||||
|
tonos, ASCII-space restoration for exotic space variants, and
|
||||||
|
removal of zero-width characters. A replacement of None
|
||||||
|
removes the codepoint."""
|
||||||
|
|
||||||
def append_provenance(path: Path, song_id: int,
|
def __init__(self, lyrics_dir: Path,
|
||||||
source: str) -> None:
|
provenance_csv: Path) -> None:
|
||||||
"""Append a provenance row for a fetched lyrics file.
|
"""Set up the fetch run.
|
||||||
|
|
||||||
The CSV file is created with the header row when missing.
|
:param lyrics_dir: The lyrics cache directory.
|
||||||
|
:param provenance_csv: The lyrics provenance CSV file.
|
||||||
|
"""
|
||||||
|
self.__lyrics_dir: Path = lyrics_dir
|
||||||
|
"""The lyrics cache directory."""
|
||||||
|
self.__provenance_csv: Path = provenance_csv
|
||||||
|
"""The lyrics provenance CSV file."""
|
||||||
|
self.__fetcher: LyricsFetcher = LyricsFetcher()
|
||||||
|
"""The fetcher of the public lyrics APIs."""
|
||||||
|
|
||||||
:param path: The lyrics provenance CSV file.
|
def run(self) -> LyricsFetchCounts:
|
||||||
:param song_id: The song ID.
|
"""Fetch the missing lyrics of every song in the store.
|
||||||
:param source: The source name of the fetched lyrics.
|
|
||||||
:return: None.
|
Every song fetched or missed is reported on the standard
|
||||||
:raises OSError: When the file cannot be written.
|
error as an observable side effect.
|
||||||
"""
|
|
||||||
is_new: bool = not path.exists()
|
:return: The number of songs fetched and missed.
|
||||||
path.parent.mkdir(parents=True, exist_ok=True)
|
:raises OSError: When a cache file or the provenance CSV
|
||||||
with open(path, "a", encoding="utf-8",
|
cannot be written.
|
||||||
newline="") as file:
|
:raises sqlalchemy.exc.SQLAlchemyError: On a database
|
||||||
writer: Any = csv.writer(file)
|
error.
|
||||||
if is_new:
|
"""
|
||||||
writer.writerow(PROVENANCE_FIELDS)
|
fetched: int = 0
|
||||||
writer.writerow([song_id, source, "api-fetch",
|
missed: int = 0
|
||||||
datetime.date.today().isoformat(), ""])
|
session: Session = ds.get_db()
|
||||||
|
try:
|
||||||
|
song: Song
|
||||||
|
for song in session.scalars(
|
||||||
|
sa.select(Song).order_by(Song.id)):
|
||||||
|
if (self.__lyrics_dir
|
||||||
|
/ f"{song.id}.txt").exists():
|
||||||
|
continue
|
||||||
|
if self.__fetch_one(session, song):
|
||||||
|
fetched += 1
|
||||||
|
else:
|
||||||
|
missed += 1
|
||||||
|
finally:
|
||||||
|
session.close()
|
||||||
|
return LyricsFetchCounts(fetched=fetched, missed=missed)
|
||||||
|
|
||||||
|
def __fetch_one(self, session: Session, song: Song) -> bool:
|
||||||
|
"""Fetch and save the lyrics of one song.
|
||||||
|
|
||||||
|
The song is queried by its primary-role artist name; when
|
||||||
|
every API misses and the song's full artist credit
|
||||||
|
differs from that name, the same APIs are queried again
|
||||||
|
with the artist credit.
|
||||||
|
|
||||||
|
:param session: The database session.
|
||||||
|
:param song: The song to fetch.
|
||||||
|
:return: True when a lyrics text was fetched and saved,
|
||||||
|
False when every API missed on both queries.
|
||||||
|
:raises OSError: When the cache file or the provenance
|
||||||
|
CSV cannot be written.
|
||||||
|
"""
|
||||||
|
artist: str = self.__query_artist(session, song.id)
|
||||||
|
result: tuple[str, str] | None = self.__fetcher.fetch(
|
||||||
|
artist, song.title)
|
||||||
|
if result is None and song.artist_credit != artist:
|
||||||
|
result = self.__fetcher.fetch(
|
||||||
|
song.artist_credit, song.title)
|
||||||
|
if result is None:
|
||||||
|
print(f"song {song.id} \"{song.title}\": miss",
|
||||||
|
file=sys.stderr)
|
||||||
|
return False
|
||||||
|
lyrics: str
|
||||||
|
source: str
|
||||||
|
lyrics, source = result
|
||||||
|
self.__save_lyrics(song.id, lyrics)
|
||||||
|
self.__append_provenance(song.id, source)
|
||||||
|
print(f"song {song.id} \"{song.title}\": {source}",
|
||||||
|
file=sys.stderr)
|
||||||
|
return True
|
||||||
|
|
||||||
|
@staticmethod
|
||||||
|
def __query_artist(session: Session, song_id: int) -> str:
|
||||||
|
"""Find the artist name to query the APIs with.
|
||||||
|
|
||||||
|
:param session: The database session.
|
||||||
|
:param song_id: The song ID.
|
||||||
|
:return: The name of the primary-role artist with the
|
||||||
|
lowest position.
|
||||||
|
"""
|
||||||
|
name: str | None = session.scalar(
|
||||||
|
sa.select(Artist.name)
|
||||||
|
.join(SongArtist, SongArtist.artist_id == Artist.id)
|
||||||
|
.where(SongArtist.song_id == song_id,
|
||||||
|
SongArtist.role == Role.PRIMARY)
|
||||||
|
.order_by(SongArtist.position)
|
||||||
|
.limit(1))
|
||||||
|
assert name is not None
|
||||||
|
return name
|
||||||
|
|
||||||
|
def __save_lyrics(self, song_id: int, lyrics: str) -> None:
|
||||||
|
"""Write the lyrics of a song into the cache directory.
|
||||||
|
|
||||||
|
The cache directory is created when missing.
|
||||||
|
|
||||||
|
The lyrics text is normalized with
|
||||||
|
:meth:`normalize_lyrics` before being written.
|
||||||
|
|
||||||
|
:param song_id: The song ID.
|
||||||
|
:param lyrics: The lyrics text.
|
||||||
|
:return: None.
|
||||||
|
:raises OSError: When the file cannot be written.
|
||||||
|
"""
|
||||||
|
self.__lyrics_dir.mkdir(parents=True, exist_ok=True)
|
||||||
|
(self.__lyrics_dir / f"{song_id}.txt").write_text(
|
||||||
|
self.normalize_lyrics(lyrics), encoding="utf-8")
|
||||||
|
|
||||||
|
def __append_provenance(self, song_id: int,
|
||||||
|
source: str) -> None:
|
||||||
|
"""Append a provenance row for a fetched lyrics file.
|
||||||
|
|
||||||
|
The CSV file is created with the header row when
|
||||||
|
missing.
|
||||||
|
|
||||||
|
:param song_id: The song ID.
|
||||||
|
:param source: The source name of the fetched lyrics.
|
||||||
|
:return: None.
|
||||||
|
:raises OSError: When the file cannot be written.
|
||||||
|
"""
|
||||||
|
is_new: bool = not self.__provenance_csv.exists()
|
||||||
|
self.__provenance_csv.parent.mkdir(
|
||||||
|
parents=True, exist_ok=True)
|
||||||
|
with open(self.__provenance_csv, "a", encoding="utf-8",
|
||||||
|
newline="") as file:
|
||||||
|
writer: Any = csv.writer(file)
|
||||||
|
if is_new:
|
||||||
|
writer.writerow(self.__PROVENANCE_FIELDS)
|
||||||
|
writer.writerow(
|
||||||
|
[song_id, source, "api-fetch",
|
||||||
|
datetime.date.today().isoformat(), ""])
|
||||||
|
|
||||||
|
@classmethod
|
||||||
|
def normalize_lyrics(cls, text: str) -> str:
|
||||||
|
"""Restore or remove watermark and mojibake characters.
|
||||||
|
|
||||||
|
:param text: The lyrics text as fetched from an API.
|
||||||
|
:return: The text with the codepoints of the
|
||||||
|
normalization table replaced or removed; every other
|
||||||
|
character is unchanged.
|
||||||
|
"""
|
||||||
|
return text.translate(cls.__NORMALIZATION)
|
||||||
|
|
||||||
|
|
||||||
def main(argv: list[str] | None = None) -> int:
|
def main(argv: list[str] | None = None) -> int:
|
||||||
@@ -277,44 +372,15 @@ def main(argv: list[str] | None = None) -> int:
|
|||||||
"""
|
"""
|
||||||
started: float = time.monotonic()
|
started: float = time.monotonic()
|
||||||
args: argparse.Namespace = parse_args(argv)
|
args: argparse.Namespace = parse_args(argv)
|
||||||
fetcher: LyricsFetcher = LyricsFetcher()
|
|
||||||
fetched: int = 0
|
|
||||||
missed: int = 0
|
|
||||||
session: Session = ds.get_db()
|
|
||||||
try:
|
try:
|
||||||
song: Song
|
counts: LyricsFetchCounts = LyricsFetchRunner(
|
||||||
for song in session.scalars(
|
args.lyrics_dir, args.provenance_csv).run()
|
||||||
sa.select(Song).order_by(Song.id)):
|
|
||||||
if (args.lyrics_dir / f"{song.id}.txt").exists():
|
|
||||||
continue
|
|
||||||
artist: str = query_artist(session, song.id)
|
|
||||||
result: tuple[str, str] | None = fetcher.fetch(
|
|
||||||
artist, song.title)
|
|
||||||
if result is None and song.artist_credit != artist:
|
|
||||||
result = fetcher.fetch(
|
|
||||||
song.artist_credit, song.title)
|
|
||||||
if result is None:
|
|
||||||
missed += 1
|
|
||||||
print(f"song {song.id} \"{song.title}\": miss",
|
|
||||||
file=sys.stderr)
|
|
||||||
continue
|
|
||||||
lyrics: str
|
|
||||||
source: str
|
|
||||||
lyrics, source = result
|
|
||||||
save_lyrics(args.lyrics_dir, song.id, lyrics)
|
|
||||||
append_provenance(args.provenance_csv, song.id,
|
|
||||||
source)
|
|
||||||
fetched += 1
|
|
||||||
print(f"song {song.id} \"{song.title}\": {source}",
|
|
||||||
file=sys.stderr)
|
|
||||||
except (OSError, sa.exc.SQLAlchemyError) as error:
|
except (OSError, sa.exc.SQLAlchemyError) as error:
|
||||||
print(f"error: {error}", file=sys.stderr)
|
print(f"error: {error}", file=sys.stderr)
|
||||||
return 1
|
return 1
|
||||||
finally:
|
attempted: int = counts.fetched + counts.missed
|
||||||
session.close()
|
|
||||||
attempted: int = fetched + missed
|
|
||||||
elapsed: str = format_duration(time.monotonic() - started)
|
elapsed: str = format_duration(time.monotonic() - started)
|
||||||
print(f"Done. Fetched lyrics for {fetched}/{attempted}"
|
print(f"Done. Fetched lyrics for {counts.fetched}/"
|
||||||
f" songs. {elapsed} elapsed.",
|
f"{attempted} songs. {elapsed} elapsed.",
|
||||||
file=sys.stderr)
|
file=sys.stderr)
|
||||||
return 0
|
return 0
|
||||||
|
|||||||
@@ -28,27 +28,13 @@ import time
|
|||||||
from dataclasses import asdict, dataclass
|
from dataclasses import asdict, dataclass
|
||||||
from datetime import datetime
|
from datetime import datetime
|
||||||
from pathlib import Path
|
from pathlib import Path
|
||||||
from typing import Any, Self
|
from typing import Any, ClassVar, Self
|
||||||
|
|
||||||
import anthropic
|
import anthropic
|
||||||
|
|
||||||
from ..config import get_settings
|
from ..config import get_settings
|
||||||
from ..utils import format_duration
|
from ..utils import format_duration
|
||||||
|
|
||||||
# claude-fable-5 accepts neither "temperature" nor "thinking";
|
|
||||||
# a model's entry holds exactly the extra request parameters it
|
|
||||||
# accepts.
|
|
||||||
MODELS: dict[str, dict[str, Any]] = {
|
|
||||||
"claude-sonnet-4-6": {
|
|
||||||
"temperature": 0.0,
|
|
||||||
"thinking": {"type": "disabled"},
|
|
||||||
},
|
|
||||||
"claude-fable-5": {},
|
|
||||||
}
|
|
||||||
DEFAULT_MODEL: str = "claude-sonnet-4-6"
|
|
||||||
SCRIPT_VERSION: str = "run_llm.py 3.1.0"
|
|
||||||
POLL_INTERVAL_SECONDS: float = 60.0
|
|
||||||
|
|
||||||
|
|
||||||
class InputFormatError(Exception):
|
class InputFormatError(Exception):
|
||||||
"""An error in the JSONL input file."""
|
"""An error in the JSONL input file."""
|
||||||
@@ -136,13 +122,23 @@ class BatchResult:
|
|||||||
if x.type == "text")
|
if x.type == "text")
|
||||||
return cls(id=entry.custom_id, text=text,
|
return cls(id=entry.custom_id, text=text,
|
||||||
stop_reason=message.stop_reason,
|
stop_reason=message.stop_reason,
|
||||||
usage=usage_to_dict(message.usage))
|
usage=cls.__usage_to_dict(message.usage))
|
||||||
case "errored":
|
case "errored":
|
||||||
return cls(id=entry.custom_id,
|
return cls(id=entry.custom_id,
|
||||||
error=result.error.error.type)
|
error=result.error.error.type)
|
||||||
case other:
|
case other:
|
||||||
return cls(id=entry.custom_id, error=str(other))
|
return cls(id=entry.custom_id, error=str(other))
|
||||||
|
|
||||||
|
@staticmethod
|
||||||
|
def __usage_to_dict(usage: Any) -> dict[str, Any]:
|
||||||
|
"""Convert a usage object to a plain dictionary.
|
||||||
|
|
||||||
|
:param usage: The usage object of a message.
|
||||||
|
:return: The usage as a dictionary, without null entries.
|
||||||
|
"""
|
||||||
|
return {k: v for k, v in usage.model_dump().items()
|
||||||
|
if v is not None}
|
||||||
|
|
||||||
def to_record(self) -> dict[str, Any]:
|
def to_record(self) -> dict[str, Any]:
|
||||||
"""Return this result as an archive JSONL record.
|
"""Return this result as an archive JSONL record.
|
||||||
|
|
||||||
@@ -169,10 +165,409 @@ class BatchInfo:
|
|||||||
is still processing."""
|
is still processing."""
|
||||||
|
|
||||||
|
|
||||||
|
@dataclass(frozen=True)
|
||||||
|
class ExecutionOutcome:
|
||||||
|
"""The outcome of submitting and awaiting one batch."""
|
||||||
|
|
||||||
|
batch: BatchInfo
|
||||||
|
"""The submitted batch's bookkeeping."""
|
||||||
|
results: Results
|
||||||
|
"""The batch's results, keyed by item ID."""
|
||||||
|
|
||||||
|
|
||||||
|
@dataclass(frozen=True)
|
||||||
|
class RunOutcome:
|
||||||
|
"""The outcome of one LLM definition file run."""
|
||||||
|
|
||||||
|
item_count: int
|
||||||
|
"""The number of loaded input items."""
|
||||||
|
dry_run: bool
|
||||||
|
"""Whether this was a dry run."""
|
||||||
|
dry_run_request: dict[str, Any] | None
|
||||||
|
"""The first item's preview request, for a dry run; None for
|
||||||
|
an actual run."""
|
||||||
|
failed: list[str]
|
||||||
|
"""The failed item IDs, in item order; always empty for a dry
|
||||||
|
run."""
|
||||||
|
|
||||||
|
|
||||||
|
class LLMRunner:
|
||||||
|
"""The orchestrator of one LLM definition file run."""
|
||||||
|
|
||||||
|
# claude-fable-5 accepts neither "temperature" nor "thinking";
|
||||||
|
# a model's entry holds exactly the extra request parameters
|
||||||
|
# it accepts.
|
||||||
|
MODELS: ClassVar[dict[str, dict[str, Any]]] = {
|
||||||
|
"claude-sonnet-4-6": {
|
||||||
|
"temperature": 0.0,
|
||||||
|
"thinking": {"type": "disabled"},
|
||||||
|
},
|
||||||
|
"claude-fable-5": {},
|
||||||
|
}
|
||||||
|
"""The supported model IDs and their extra request
|
||||||
|
parameters."""
|
||||||
|
DEFAULT_MODEL: ClassVar[str] = "claude-sonnet-4-6"
|
||||||
|
"""The default model ID."""
|
||||||
|
__SCRIPT_VERSION: ClassVar[str] = "run_llm.py 3.1.0"
|
||||||
|
"""The script version recorded into the archive metadata."""
|
||||||
|
__POLL_INTERVAL_SECONDS: ClassVar[float] = 60.0
|
||||||
|
"""The interval between batch status polls."""
|
||||||
|
|
||||||
|
def __init__(self, prompt: Path, input_path: Path,
|
||||||
|
archive_dir: Path, model: str, max_tokens: int,
|
||||||
|
dry_run: bool, replace: bool) -> None:
|
||||||
|
"""Set up the run of one LLM definition file.
|
||||||
|
|
||||||
|
:param prompt: The prompt definition file, used as the
|
||||||
|
system prompt.
|
||||||
|
:param input_path: The JSONL input file with "id" and
|
||||||
|
"content".
|
||||||
|
:param archive_dir: The destination archive directory.
|
||||||
|
:param model: The model ID, a key of :attr:`MODELS`.
|
||||||
|
:param max_tokens: The maximum output tokens per request.
|
||||||
|
:param dry_run: Whether to validate and archive without
|
||||||
|
calling the API.
|
||||||
|
:param replace: Whether to replace an already existing
|
||||||
|
archive directory.
|
||||||
|
"""
|
||||||
|
self.__prompt: Path = prompt
|
||||||
|
"""The prompt definition file."""
|
||||||
|
self.__input: Path = input_path
|
||||||
|
"""The JSONL input file."""
|
||||||
|
self.__archive_dir: Path = archive_dir
|
||||||
|
"""The destination archive directory."""
|
||||||
|
self.__model: str = model
|
||||||
|
"""The model ID."""
|
||||||
|
self.__max_tokens: int = max_tokens
|
||||||
|
"""The maximum output tokens per request."""
|
||||||
|
self.__dry_run: bool = dry_run
|
||||||
|
"""Whether to validate and archive without calling the
|
||||||
|
API."""
|
||||||
|
self.__replace: bool = replace
|
||||||
|
"""Whether to replace an already existing archive
|
||||||
|
directory."""
|
||||||
|
|
||||||
|
def run(self) -> RunOutcome:
|
||||||
|
"""Load the input, archive the prompt, and run the batch.
|
||||||
|
|
||||||
|
Always writes ``prompt.md`` and ``meta.json`` into the
|
||||||
|
archive directory. A dry run stops there, previewing the
|
||||||
|
first item's request; an actual run also submits the
|
||||||
|
batch, awaits it, and writes ``output.jsonl``.
|
||||||
|
|
||||||
|
:return: The outcome of the run.
|
||||||
|
:raises InputFormatError: When the input file is
|
||||||
|
malformed.
|
||||||
|
:raises OSError: When the input or prompt file cannot be
|
||||||
|
read, the archive directory already exists without
|
||||||
|
``replace``, or an output file cannot be written.
|
||||||
|
"""
|
||||||
|
items: list[InputItem] = self.__load_items()
|
||||||
|
prompt_text: str = self.__prompt.read_text(encoding="utf-8")
|
||||||
|
archive_dir: Path = self.__create_archive_dir()
|
||||||
|
(archive_dir / "prompt.md").write_bytes(
|
||||||
|
self.__prompt.read_bytes())
|
||||||
|
meta: dict[str, Any] = self.__build_meta(items)
|
||||||
|
meta_path: Path = archive_dir / "meta.json"
|
||||||
|
if self.__dry_run:
|
||||||
|
self.__write_json(meta_path, meta)
|
||||||
|
request: dict[str, Any] = self.__build_request(
|
||||||
|
items[0], prompt_text)
|
||||||
|
return RunOutcome(
|
||||||
|
item_count=len(items), dry_run=True,
|
||||||
|
dry_run_request=request, failed=[])
|
||||||
|
client: anthropic.Anthropic = anthropic.Anthropic(
|
||||||
|
api_key=get_settings().ANTHROPIC_API_KEY)
|
||||||
|
outcome: ExecutionOutcome = self.__execute_run(
|
||||||
|
client, items, prompt_text)
|
||||||
|
item_ids: list[str] = [x.id for x in items]
|
||||||
|
self.__write_jsonl(
|
||||||
|
archive_dir / "output.jsonl",
|
||||||
|
[outcome.results[x].to_record() for x in item_ids
|
||||||
|
if x in outcome.results])
|
||||||
|
meta["batch"] = outcome.batch
|
||||||
|
meta["usage"] = self.__sum_usage(outcome.results)
|
||||||
|
self.__write_meta(meta_path, meta)
|
||||||
|
failed: list[str] = self.__find_failures(
|
||||||
|
item_ids, outcome.results)
|
||||||
|
return RunOutcome(
|
||||||
|
item_count=len(items), dry_run=False,
|
||||||
|
dry_run_request=None, failed=failed)
|
||||||
|
|
||||||
|
def __load_items(self) -> list[InputItem]:
|
||||||
|
"""Load and validate the JSONL input items.
|
||||||
|
|
||||||
|
:return: The input items, in file order.
|
||||||
|
:raises InputFormatError: When a line is malformed, an ID
|
||||||
|
is duplicated, or the file contains no item.
|
||||||
|
:raises OSError: When the file cannot be read.
|
||||||
|
"""
|
||||||
|
items: list[InputItem] = []
|
||||||
|
seen: set[str] = set()
|
||||||
|
with open(self.__input, encoding="utf-8") as file:
|
||||||
|
for number, line in enumerate(file, start=1):
|
||||||
|
if line.strip() == "":
|
||||||
|
continue
|
||||||
|
data: Any
|
||||||
|
try:
|
||||||
|
data = json.loads(line)
|
||||||
|
except json.JSONDecodeError as error:
|
||||||
|
raise InputFormatError(
|
||||||
|
f"{self.__input}: line {number}: malformed"
|
||||||
|
f" JSON: {error}")
|
||||||
|
item: InputItem = InputItem.get_instance(
|
||||||
|
data, self.__input, number)
|
||||||
|
if item.id in seen:
|
||||||
|
raise InputFormatError(
|
||||||
|
f"{self.__input}: line {number}:"
|
||||||
|
f" duplicated ID \"{item.id}\"")
|
||||||
|
seen.add(item.id)
|
||||||
|
items.append(item)
|
||||||
|
if len(items) == 0:
|
||||||
|
raise InputFormatError(f"{self.__input}: no input items")
|
||||||
|
return items
|
||||||
|
|
||||||
|
def __create_archive_dir(self) -> Path:
|
||||||
|
"""Create the archive directory.
|
||||||
|
|
||||||
|
Only this directory is ever created or removed; no other
|
||||||
|
directory is ever touched.
|
||||||
|
|
||||||
|
:return: The created archive directory.
|
||||||
|
:raises FileExistsError: When the archive directory
|
||||||
|
already exists and ``replace`` is False.
|
||||||
|
"""
|
||||||
|
if self.__archive_dir.exists():
|
||||||
|
if not self.__replace:
|
||||||
|
raise FileExistsError(
|
||||||
|
f"{self.__archive_dir} already exists; pass"
|
||||||
|
" --replace to replace it")
|
||||||
|
shutil.rmtree(self.__archive_dir)
|
||||||
|
self.__archive_dir.mkdir(parents=True)
|
||||||
|
return self.__archive_dir
|
||||||
|
|
||||||
|
def __build_meta(self, items: list[InputItem]) -> dict[str, Any]:
|
||||||
|
"""Build the initial archive metadata.
|
||||||
|
|
||||||
|
:param items: The loaded input items.
|
||||||
|
:return: The metadata, "batch" and "usage" not yet filled
|
||||||
|
in for an actual run.
|
||||||
|
"""
|
||||||
|
return {
|
||||||
|
"script_version": self.__SCRIPT_VERSION,
|
||||||
|
"model": self.__model,
|
||||||
|
"temperature": self.MODELS[self.__model].get(
|
||||||
|
"temperature"),
|
||||||
|
"thinking": self.MODELS[self.__model].get("thinking"),
|
||||||
|
"max_tokens": self.__max_tokens,
|
||||||
|
"prompt_path": str(self.__prompt),
|
||||||
|
"prompt_sha256": self.__sha256_of(self.__prompt),
|
||||||
|
"input_path": str(self.__input),
|
||||||
|
"input_sha256": self.__sha256_of(self.__input),
|
||||||
|
"item_count": len(items),
|
||||||
|
"dry_run": self.__dry_run,
|
||||||
|
"started_at": self.__now_iso(),
|
||||||
|
"batch": None,
|
||||||
|
"usage": {},
|
||||||
|
}
|
||||||
|
|
||||||
|
def __build_request(self, item: InputItem, system_prompt: str) \
|
||||||
|
-> dict[str, Any]:
|
||||||
|
"""Build one Message Batches request for an input item.
|
||||||
|
|
||||||
|
:param item: The input item.
|
||||||
|
:param system_prompt: The system prompt text.
|
||||||
|
:return: The batch request with "custom_id" and "params".
|
||||||
|
"""
|
||||||
|
return {
|
||||||
|
"custom_id": item.id,
|
||||||
|
"params": {
|
||||||
|
"model": self.__model,
|
||||||
|
"max_tokens": self.__max_tokens,
|
||||||
|
**self.MODELS[self.__model],
|
||||||
|
"system": system_prompt,
|
||||||
|
"messages": [
|
||||||
|
{"role": "user", "content": item.content},
|
||||||
|
],
|
||||||
|
},
|
||||||
|
}
|
||||||
|
|
||||||
|
def __execute_run(
|
||||||
|
self, client: anthropic.Anthropic,
|
||||||
|
items: list[InputItem], system_prompt: str) \
|
||||||
|
-> ExecutionOutcome:
|
||||||
|
"""Submit the batch of this run and await its results.
|
||||||
|
|
||||||
|
:param client: The Anthropic client.
|
||||||
|
:param items: The input items.
|
||||||
|
:param system_prompt: The system prompt text.
|
||||||
|
:return: The submitted batch's bookkeeping and its
|
||||||
|
results.
|
||||||
|
"""
|
||||||
|
requests: list[dict[str, Any]] = [
|
||||||
|
self.__build_request(x, system_prompt) for x in items]
|
||||||
|
info: BatchInfo = BatchInfo(
|
||||||
|
batch_id=self.__submit_batch(client, requests),
|
||||||
|
submitted_at=self.__now_iso())
|
||||||
|
print(f"submitted batch {info.batch_id}", file=sys.stderr)
|
||||||
|
batches: dict[str, Any] = self.__poll_batches(
|
||||||
|
client, [info.batch_id])
|
||||||
|
info.ended_at = batches[info.batch_id].ended_at.isoformat()
|
||||||
|
results: Results = self.__collect_results(
|
||||||
|
client, info.batch_id)
|
||||||
|
return ExecutionOutcome(batch=info, results=results)
|
||||||
|
|
||||||
|
@staticmethod
|
||||||
|
def __submit_batch(client: anthropic.Anthropic,
|
||||||
|
requests: list[dict[str, Any]]) -> str:
|
||||||
|
"""Submit one message batch.
|
||||||
|
|
||||||
|
:param client: The Anthropic client.
|
||||||
|
:param requests: The batch requests.
|
||||||
|
:return: The batch ID.
|
||||||
|
"""
|
||||||
|
return client.messages.batches.create(requests=requests).id
|
||||||
|
|
||||||
|
@classmethod
|
||||||
|
def __poll_batches(cls, client: anthropic.Anthropic,
|
||||||
|
batch_ids: list[str]) -> dict[str, Any]:
|
||||||
|
"""Poll the batches until every one of them has ended.
|
||||||
|
|
||||||
|
Progress is printed to the standard error every poll.
|
||||||
|
|
||||||
|
:param client: The Anthropic client.
|
||||||
|
:param batch_ids: The batch IDs to poll.
|
||||||
|
:return: The final batch object of each batch, keyed by
|
||||||
|
batch ID.
|
||||||
|
"""
|
||||||
|
while True:
|
||||||
|
batches: dict[str, Any] = {
|
||||||
|
x: client.messages.batches.retrieve(x)
|
||||||
|
for x in batch_ids}
|
||||||
|
pending: list[str] = [
|
||||||
|
x for x in batch_ids
|
||||||
|
if batches[x].processing_status != "ended"]
|
||||||
|
for batch_id in batch_ids:
|
||||||
|
status: str = batches[batch_id].processing_status
|
||||||
|
print(f"batch {batch_id}: {status}", file=sys.stderr)
|
||||||
|
if len(pending) == 0:
|
||||||
|
return batches
|
||||||
|
time.sleep(cls.__POLL_INTERVAL_SECONDS)
|
||||||
|
|
||||||
|
@staticmethod
|
||||||
|
def __collect_results(client: anthropic.Anthropic,
|
||||||
|
batch_id: str) -> Results:
|
||||||
|
"""Collect the results of an ended batch.
|
||||||
|
|
||||||
|
:param client: The Anthropic client.
|
||||||
|
:param batch_id: The batch ID.
|
||||||
|
:return: The result records, keyed by custom ID.
|
||||||
|
"""
|
||||||
|
results: Results = {}
|
||||||
|
for entry in client.messages.batches.results(batch_id):
|
||||||
|
results[entry.custom_id] = BatchResult.get_instance(
|
||||||
|
entry)
|
||||||
|
return results
|
||||||
|
|
||||||
|
@staticmethod
|
||||||
|
def __find_failures(item_ids: list[str],
|
||||||
|
results: Results) -> list[str]:
|
||||||
|
"""Find the item IDs that failed in a result set.
|
||||||
|
|
||||||
|
An item failed when it is missing from the results or when
|
||||||
|
its record is a failure.
|
||||||
|
|
||||||
|
:param item_ids: The item IDs to check, in order.
|
||||||
|
:param results: The result records, keyed by item ID.
|
||||||
|
:return: The failed item IDs, in the given order.
|
||||||
|
"""
|
||||||
|
return [x for x in item_ids
|
||||||
|
if x not in results or results[x].is_failure]
|
||||||
|
|
||||||
|
@staticmethod
|
||||||
|
def __sum_usage(results: Results) -> dict[str, int]:
|
||||||
|
"""Sum the token usage of every succeeded result.
|
||||||
|
|
||||||
|
:param results: The result records, keyed by item ID.
|
||||||
|
:return: The summed integer usage fields.
|
||||||
|
"""
|
||||||
|
totals: dict[str, int] = {}
|
||||||
|
for result in results.values():
|
||||||
|
if result.usage is None:
|
||||||
|
continue
|
||||||
|
for key, value in result.usage.items():
|
||||||
|
if isinstance(value, int):
|
||||||
|
totals[key] = totals.get(key, 0) + value
|
||||||
|
return totals
|
||||||
|
|
||||||
|
@staticmethod
|
||||||
|
def __write_jsonl(path: Path,
|
||||||
|
records: list[dict[str, Any]]) -> None:
|
||||||
|
"""Write records to a file as JSON Lines.
|
||||||
|
|
||||||
|
:param path: The path of the file to write.
|
||||||
|
:param records: The records, one per line.
|
||||||
|
:return: None.
|
||||||
|
:raises OSError: When the file cannot be written.
|
||||||
|
"""
|
||||||
|
with open(path, "w", encoding="utf-8") as file:
|
||||||
|
for record in records:
|
||||||
|
file.write(
|
||||||
|
json.dumps(record, ensure_ascii=False) + "\n")
|
||||||
|
|
||||||
|
@staticmethod
|
||||||
|
def __write_json(path: Path, data: dict[str, Any]) -> None:
|
||||||
|
"""Write data to a file as pretty-printed JSON.
|
||||||
|
|
||||||
|
:param path: The path of the file to write.
|
||||||
|
:param data: The data to write.
|
||||||
|
:return: None.
|
||||||
|
:raises OSError: When the file cannot be written.
|
||||||
|
"""
|
||||||
|
path.write_text(
|
||||||
|
json.dumps(data, ensure_ascii=False, indent=2) + "\n",
|
||||||
|
encoding="utf-8")
|
||||||
|
|
||||||
|
@classmethod
|
||||||
|
def __write_meta(cls, path: Path, meta: dict[str, Any]) -> None:
|
||||||
|
"""Write the metadata to the ``meta.json`` file.
|
||||||
|
|
||||||
|
The ``BatchInfo`` value under "batch" is written as a
|
||||||
|
plain JSON object.
|
||||||
|
|
||||||
|
:param path: The path of the ``meta.json`` file.
|
||||||
|
:param meta: The metadata to write.
|
||||||
|
:return: None.
|
||||||
|
:raises OSError: When the file cannot be written.
|
||||||
|
"""
|
||||||
|
cls.__write_json(
|
||||||
|
path, {**meta, "batch": asdict(meta["batch"])})
|
||||||
|
|
||||||
|
@staticmethod
|
||||||
|
def __sha256_of(path: Path) -> str:
|
||||||
|
"""Calculate the SHA-256 digest of a file.
|
||||||
|
|
||||||
|
:param path: The path of the file.
|
||||||
|
:return: The hexadecimal SHA-256 digest.
|
||||||
|
"""
|
||||||
|
with open(path, "rb") as file:
|
||||||
|
return hashlib.file_digest(file, "sha256").hexdigest()
|
||||||
|
|
||||||
|
@staticmethod
|
||||||
|
def __now_iso() -> str:
|
||||||
|
"""Return the current local time in ISO 8601 format.
|
||||||
|
|
||||||
|
:return: The current local time with the timezone offset.
|
||||||
|
"""
|
||||||
|
return datetime.now().astimezone().isoformat(
|
||||||
|
timespec="seconds")
|
||||||
|
|
||||||
|
|
||||||
def parse_args(argv: list[str] | None) -> argparse.Namespace:
|
def parse_args(argv: list[str] | None) -> argparse.Namespace:
|
||||||
"""Parse the command-line arguments.
|
"""Parse the command-line arguments.
|
||||||
|
|
||||||
:param argv: The command-line arguments, or None for ``sys.argv``.
|
:param argv: The command-line arguments, or None for
|
||||||
|
``sys.argv``.
|
||||||
:return: The parsed arguments.
|
:return: The parsed arguments.
|
||||||
"""
|
"""
|
||||||
parser: argparse.ArgumentParser = argparse.ArgumentParser(
|
parser: argparse.ArgumentParser = argparse.ArgumentParser(
|
||||||
@@ -180,7 +575,8 @@ def parse_args(argv: list[str] | None) -> argparse.Namespace:
|
|||||||
" and archive the result.")
|
" and archive the result.")
|
||||||
parser.add_argument(
|
parser.add_argument(
|
||||||
"prompt", type=Path,
|
"prompt", type=Path,
|
||||||
help="the prompt definition file, used as the system prompt")
|
help="the prompt definition file, used as the system"
|
||||||
|
" prompt")
|
||||||
parser.add_argument(
|
parser.add_argument(
|
||||||
"input", type=Path,
|
"input", type=Path,
|
||||||
help="the JSONL input file with \"id\" and \"content\"")
|
help="the JSONL input file with \"id\" and \"content\"")
|
||||||
@@ -188,11 +584,13 @@ def parse_args(argv: list[str] | None) -> argparse.Namespace:
|
|||||||
"archive_dir", type=Path,
|
"archive_dir", type=Path,
|
||||||
help="the destination archive directory")
|
help="the destination archive directory")
|
||||||
parser.add_argument(
|
parser.add_argument(
|
||||||
"--model", choices=sorted(MODELS), default=DEFAULT_MODEL,
|
"--model", choices=sorted(LLMRunner.MODELS),
|
||||||
help=f"the model ID (default {DEFAULT_MODEL})")
|
default=LLMRunner.DEFAULT_MODEL,
|
||||||
|
help=f"the model ID (default {LLMRunner.DEFAULT_MODEL})")
|
||||||
parser.add_argument(
|
parser.add_argument(
|
||||||
"--max-tokens", type=int, default=2048,
|
"--max-tokens", type=int, default=2048,
|
||||||
help="the maximum output tokens per request (default 2048)")
|
help="the maximum output tokens per request (default"
|
||||||
|
" 2048)")
|
||||||
parser.add_argument(
|
parser.add_argument(
|
||||||
"--dry-run", action="store_true",
|
"--dry-run", action="store_true",
|
||||||
help="validate and archive without calling the API")
|
help="validate and archive without calling the API")
|
||||||
@@ -202,327 +600,30 @@ def parse_args(argv: list[str] | None) -> argparse.Namespace:
|
|||||||
return parser.parse_args(argv)
|
return parser.parse_args(argv)
|
||||||
|
|
||||||
|
|
||||||
def load_items(path: Path) -> list[InputItem]:
|
|
||||||
"""Load and validate the JSONL input items.
|
|
||||||
|
|
||||||
:param path: The path of the JSONL input file.
|
|
||||||
:return: The input items, in file order.
|
|
||||||
:raises InputFormatError: When a line is malformed, an ID is
|
|
||||||
duplicated, or the file contains no item.
|
|
||||||
:raises OSError: When the file cannot be read.
|
|
||||||
"""
|
|
||||||
items: list[InputItem] = []
|
|
||||||
seen: set[str] = set()
|
|
||||||
with open(path, encoding="utf-8") as file:
|
|
||||||
for number, line in enumerate(file, start=1):
|
|
||||||
if line.strip() == "":
|
|
||||||
continue
|
|
||||||
try:
|
|
||||||
data: Any = json.loads(line)
|
|
||||||
except json.JSONDecodeError as error:
|
|
||||||
raise InputFormatError(
|
|
||||||
f"{path}: line {number}: malformed JSON: {error}")
|
|
||||||
item: InputItem = InputItem.get_instance(
|
|
||||||
data, path, number)
|
|
||||||
if item.id in seen:
|
|
||||||
raise InputFormatError(
|
|
||||||
f"{path}: line {number}: duplicated ID"
|
|
||||||
f" \"{item.id}\"")
|
|
||||||
seen.add(item.id)
|
|
||||||
items.append(item)
|
|
||||||
if len(items) == 0:
|
|
||||||
raise InputFormatError(f"{path}: no input items")
|
|
||||||
return items
|
|
||||||
|
|
||||||
|
|
||||||
def build_request(item: InputItem, system_prompt: str,
|
|
||||||
max_tokens: int, model: str) -> dict[str, Any]:
|
|
||||||
"""Build one Message Batches request for an input item.
|
|
||||||
|
|
||||||
:param item: The input item.
|
|
||||||
:param system_prompt: The system prompt text.
|
|
||||||
:param max_tokens: The maximum output tokens.
|
|
||||||
:param model: The model ID, a key of ``MODELS``.
|
|
||||||
:return: The batch request with "custom_id" and "params".
|
|
||||||
"""
|
|
||||||
return {
|
|
||||||
"custom_id": item.id,
|
|
||||||
"params": {
|
|
||||||
"model": model,
|
|
||||||
"max_tokens": max_tokens,
|
|
||||||
**MODELS[model],
|
|
||||||
"system": system_prompt,
|
|
||||||
"messages": [
|
|
||||||
{"role": "user", "content": item.content},
|
|
||||||
],
|
|
||||||
},
|
|
||||||
}
|
|
||||||
|
|
||||||
|
|
||||||
def submit_batch(client: anthropic.Anthropic,
|
|
||||||
requests: list[dict[str, Any]]) -> str:
|
|
||||||
"""Submit one message batch.
|
|
||||||
|
|
||||||
:param client: The Anthropic client.
|
|
||||||
:param requests: The batch requests.
|
|
||||||
:return: The batch ID.
|
|
||||||
"""
|
|
||||||
return client.messages.batches.create(requests=requests).id
|
|
||||||
|
|
||||||
|
|
||||||
def poll_batches(client: anthropic.Anthropic,
|
|
||||||
batch_ids: list[str]) -> dict[str, Any]:
|
|
||||||
"""Poll the batches until every one of them has ended.
|
|
||||||
|
|
||||||
Progress is printed to the standard error every poll.
|
|
||||||
|
|
||||||
:param client: The Anthropic client.
|
|
||||||
:param batch_ids: The batch IDs to poll.
|
|
||||||
:return: The final batch object of each batch, keyed by batch ID.
|
|
||||||
"""
|
|
||||||
while True:
|
|
||||||
batches: dict[str, Any] = {
|
|
||||||
x: client.messages.batches.retrieve(x) for x in batch_ids}
|
|
||||||
pending: list[str] = [
|
|
||||||
x for x in batch_ids
|
|
||||||
if batches[x].processing_status != "ended"]
|
|
||||||
for batch_id in batch_ids:
|
|
||||||
status: str = batches[batch_id].processing_status
|
|
||||||
print(f"batch {batch_id}: {status}", file=sys.stderr)
|
|
||||||
if len(pending) == 0:
|
|
||||||
return batches
|
|
||||||
time.sleep(POLL_INTERVAL_SECONDS)
|
|
||||||
|
|
||||||
|
|
||||||
def usage_to_dict(usage: Any) -> dict[str, Any]:
|
|
||||||
"""Convert a usage object to a plain dictionary.
|
|
||||||
|
|
||||||
:param usage: The usage object of a message.
|
|
||||||
:return: The usage as a dictionary, without null entries.
|
|
||||||
"""
|
|
||||||
return {k: v for k, v in usage.model_dump().items()
|
|
||||||
if v is not None}
|
|
||||||
|
|
||||||
|
|
||||||
def sum_usage(results: Results) -> dict[str, int]:
|
|
||||||
"""Sum the token usage of every succeeded result.
|
|
||||||
|
|
||||||
:param results: The result records, keyed by item ID.
|
|
||||||
:return: The summed integer usage fields.
|
|
||||||
"""
|
|
||||||
totals: dict[str, int] = {}
|
|
||||||
for result in results.values():
|
|
||||||
if result.usage is None:
|
|
||||||
continue
|
|
||||||
for key, value in result.usage.items():
|
|
||||||
if isinstance(value, int):
|
|
||||||
totals[key] = totals.get(key, 0) + value
|
|
||||||
return totals
|
|
||||||
|
|
||||||
|
|
||||||
def collect_results(client: anthropic.Anthropic,
|
|
||||||
batch_id: str) -> Results:
|
|
||||||
"""Collect the results of an ended batch.
|
|
||||||
|
|
||||||
:param client: The Anthropic client.
|
|
||||||
:param batch_id: The batch ID.
|
|
||||||
:return: The result records, keyed by custom ID.
|
|
||||||
"""
|
|
||||||
results: Results = {}
|
|
||||||
for entry in client.messages.batches.results(batch_id):
|
|
||||||
results[entry.custom_id] = BatchResult.get_instance(entry)
|
|
||||||
return results
|
|
||||||
|
|
||||||
|
|
||||||
def find_failures(item_ids: list[str],
|
|
||||||
results: Results) -> list[str]:
|
|
||||||
"""Find the item IDs that failed in a result set.
|
|
||||||
|
|
||||||
An item failed when it is missing from the results or when its
|
|
||||||
record is a failure.
|
|
||||||
|
|
||||||
:param item_ids: The item IDs to check, in order.
|
|
||||||
:param results: The result records, keyed by item ID.
|
|
||||||
:return: The failed item IDs, in the given order.
|
|
||||||
"""
|
|
||||||
return [x for x in item_ids
|
|
||||||
if x not in results or results[x].is_failure]
|
|
||||||
|
|
||||||
|
|
||||||
def create_archive_dir(directory: Path, replace: bool) -> Path:
|
|
||||||
"""Create the archive directory.
|
|
||||||
|
|
||||||
Only this directory is ever created or removed; no other
|
|
||||||
directory is ever touched.
|
|
||||||
|
|
||||||
:param directory: The destination archive directory.
|
|
||||||
:param replace: Whether to remove an already existing archive
|
|
||||||
directory before creating it.
|
|
||||||
:return: The created archive directory.
|
|
||||||
:raises FileExistsError: When the archive directory already
|
|
||||||
exists and ``replace`` is False.
|
|
||||||
"""
|
|
||||||
if directory.exists():
|
|
||||||
if not replace:
|
|
||||||
raise FileExistsError(
|
|
||||||
f"{directory} already exists; pass --replace to"
|
|
||||||
" replace it")
|
|
||||||
shutil.rmtree(directory)
|
|
||||||
directory.mkdir(parents=True)
|
|
||||||
return directory
|
|
||||||
|
|
||||||
|
|
||||||
def write_jsonl(path: Path, records: list[dict[str, Any]]) -> None:
|
|
||||||
"""Write records to a file as JSON Lines.
|
|
||||||
|
|
||||||
:param path: The path of the file to write.
|
|
||||||
:param records: The records, one per line.
|
|
||||||
:return: None.
|
|
||||||
"""
|
|
||||||
with open(path, "w", encoding="utf-8") as file:
|
|
||||||
for record in records:
|
|
||||||
file.write(json.dumps(record, ensure_ascii=False) + "\n")
|
|
||||||
|
|
||||||
|
|
||||||
def write_json(path: Path, data: dict[str, Any]) -> None:
|
|
||||||
"""Write data to a file as pretty-printed JSON.
|
|
||||||
|
|
||||||
:param path: The path of the file to write.
|
|
||||||
:param data: The data to write.
|
|
||||||
:return: None.
|
|
||||||
"""
|
|
||||||
path.write_text(
|
|
||||||
json.dumps(data, ensure_ascii=False, indent=2) + "\n",
|
|
||||||
encoding="utf-8")
|
|
||||||
|
|
||||||
|
|
||||||
def write_meta(path: Path, meta: dict[str, Any]) -> None:
|
|
||||||
"""Write the metadata to the ``meta.json`` file.
|
|
||||||
|
|
||||||
The ``BatchInfo`` value under ``batch`` is written as a plain
|
|
||||||
JSON object.
|
|
||||||
|
|
||||||
:param path: The path of the ``meta.json`` file.
|
|
||||||
:param meta: The metadata to write.
|
|
||||||
:return: None.
|
|
||||||
"""
|
|
||||||
write_json(path, {**meta, "batch": asdict(meta["batch"])})
|
|
||||||
|
|
||||||
|
|
||||||
def sha256_of(path: Path) -> str:
|
|
||||||
"""Calculate the SHA-256 digest of a file.
|
|
||||||
|
|
||||||
:param path: The path of the file.
|
|
||||||
:return: The hexadecimal SHA-256 digest.
|
|
||||||
"""
|
|
||||||
with open(path, "rb") as file:
|
|
||||||
return hashlib.file_digest(file, "sha256").hexdigest()
|
|
||||||
|
|
||||||
|
|
||||||
def now_iso() -> str:
|
|
||||||
"""Return the current local time in ISO 8601 format.
|
|
||||||
|
|
||||||
:return: The current local time with the timezone offset.
|
|
||||||
"""
|
|
||||||
return datetime.now().astimezone().isoformat(timespec="seconds")
|
|
||||||
|
|
||||||
|
|
||||||
def execute_run(
|
|
||||||
client: anthropic.Anthropic, items: list[InputItem],
|
|
||||||
system_prompt: str, max_tokens: int, model: str,
|
|
||||||
meta: dict[str, Any],
|
|
||||||
) -> Results:
|
|
||||||
"""Submit the batch of this run and await its results.
|
|
||||||
|
|
||||||
The batch ID and timestamps are recorded into the metadata as an
|
|
||||||
observable side effect.
|
|
||||||
|
|
||||||
:param client: The Anthropic client.
|
|
||||||
:param items: The input items.
|
|
||||||
:param system_prompt: The system prompt text.
|
|
||||||
:param max_tokens: The maximum output tokens per request.
|
|
||||||
:param model: The model ID, a key of ``MODELS``.
|
|
||||||
:param meta: The metadata to record the batch bookkeeping into.
|
|
||||||
:return: The results of this run, keyed by item ID.
|
|
||||||
"""
|
|
||||||
requests: list[dict[str, Any]] = [
|
|
||||||
build_request(x, system_prompt, max_tokens, model)
|
|
||||||
for x in items]
|
|
||||||
info: BatchInfo = BatchInfo(
|
|
||||||
batch_id=submit_batch(client, requests),
|
|
||||||
submitted_at=now_iso())
|
|
||||||
meta["batch"] = info
|
|
||||||
print(f"submitted batch {info.batch_id}", file=sys.stderr)
|
|
||||||
batches: dict[str, Any] = poll_batches(client, [info.batch_id])
|
|
||||||
info.ended_at = batches[info.batch_id].ended_at.isoformat()
|
|
||||||
return collect_results(client, info.batch_id)
|
|
||||||
|
|
||||||
|
|
||||||
def main(argv: list[str] | None = None) -> int:
|
def main(argv: list[str] | None = None) -> int:
|
||||||
"""Run one LLM definition file against one input and archive it.
|
"""Run one LLM definition file against one input and archive it.
|
||||||
|
|
||||||
:param argv: The command-line arguments, or None for ``sys.argv``.
|
:param argv: The command-line arguments, or None for
|
||||||
|
``sys.argv``.
|
||||||
:return: The exit status: 0 on success, non-zero on failure.
|
:return: The exit status: 0 on success, non-zero on failure.
|
||||||
"""
|
"""
|
||||||
started: float = time.monotonic()
|
started: float = time.monotonic()
|
||||||
args: argparse.Namespace = parse_args(argv)
|
args: argparse.Namespace = parse_args(argv)
|
||||||
try:
|
try:
|
||||||
items: list[InputItem] = load_items(args.input)
|
outcome: RunOutcome = LLMRunner(
|
||||||
prompt_text: str = args.prompt.read_text(encoding="utf-8")
|
args.prompt, args.input, args.archive_dir, args.model,
|
||||||
except (OSError, InputFormatError) as error:
|
args.max_tokens, args.dry_run, args.replace).run()
|
||||||
|
except (InputFormatError, OSError) as error:
|
||||||
print(f"error: {error}", file=sys.stderr)
|
print(f"error: {error}", file=sys.stderr)
|
||||||
return 1
|
return 1
|
||||||
try:
|
if not outcome.dry_run and len(outcome.failed) > 0:
|
||||||
archive_dir: Path = create_archive_dir(
|
print(f"error: failed items: {', '.join(outcome.failed)}",
|
||||||
args.archive_dir, args.replace)
|
|
||||||
except FileExistsError as error:
|
|
||||||
print(f"error: {error}", file=sys.stderr)
|
|
||||||
return 1
|
|
||||||
meta_path: Path = archive_dir / "meta.json"
|
|
||||||
(archive_dir / "prompt.md").write_bytes(args.prompt.read_bytes())
|
|
||||||
meta: dict[str, Any] = {
|
|
||||||
"script_version": SCRIPT_VERSION,
|
|
||||||
"model": args.model,
|
|
||||||
"temperature": MODELS[args.model].get("temperature"),
|
|
||||||
"thinking": MODELS[args.model].get("thinking"),
|
|
||||||
"max_tokens": args.max_tokens,
|
|
||||||
"prompt_path": str(args.prompt),
|
|
||||||
"prompt_sha256": sha256_of(args.prompt),
|
|
||||||
"input_path": str(args.input),
|
|
||||||
"input_sha256": sha256_of(args.input),
|
|
||||||
"item_count": len(items),
|
|
||||||
"dry_run": args.dry_run,
|
|
||||||
"started_at": now_iso(),
|
|
||||||
"batch": None,
|
|
||||||
"usage": {},
|
|
||||||
}
|
|
||||||
if args.dry_run:
|
|
||||||
write_json(meta_path, meta)
|
|
||||||
print(json.dumps(
|
|
||||||
build_request(items[0], prompt_text, args.max_tokens,
|
|
||||||
args.model),
|
|
||||||
ensure_ascii=False, indent=2))
|
|
||||||
elapsed: str = format_duration(time.monotonic() - started)
|
|
||||||
print(f"Done. {len(items)} jobs finished."
|
|
||||||
f" {elapsed} elapsed.", file=sys.stderr)
|
|
||||||
return 0
|
|
||||||
client: anthropic.Anthropic = anthropic.Anthropic(
|
|
||||||
api_key=get_settings().ANTHROPIC_API_KEY)
|
|
||||||
results: Results = execute_run(
|
|
||||||
client, items, prompt_text, args.max_tokens, args.model,
|
|
||||||
meta)
|
|
||||||
item_ids: list[str] = [x.id for x in items]
|
|
||||||
write_jsonl(
|
|
||||||
archive_dir / "output.jsonl",
|
|
||||||
[results[x].to_record() for x in item_ids if x in results])
|
|
||||||
meta["usage"] = sum_usage(results)
|
|
||||||
write_meta(meta_path, meta)
|
|
||||||
failed: list[str] = find_failures(item_ids, results)
|
|
||||||
if len(failed) > 0:
|
|
||||||
print(f"error: failed items: {', '.join(failed)}",
|
|
||||||
file=sys.stderr)
|
file=sys.stderr)
|
||||||
return 1
|
return 1
|
||||||
|
if outcome.dry_run:
|
||||||
|
print(json.dumps(
|
||||||
|
outcome.dry_run_request, ensure_ascii=False, indent=2))
|
||||||
elapsed: str = format_duration(time.monotonic() - started)
|
elapsed: str = format_duration(time.monotonic() - started)
|
||||||
print(f"Done. {len(items)} jobs finished."
|
print(f"Done. {outcome.item_count} jobs finished."
|
||||||
f" {elapsed} elapsed.", file=sys.stderr)
|
f" {elapsed} elapsed.", file=sys.stderr)
|
||||||
return 0
|
return 0
|
||||||
|
|||||||
File diff suppressed because it is too large
Load Diff
@@ -6,49 +6,16 @@ r"""The majority tally of the three coding runs.
|
|||||||
|
|
||||||
Settles the coding step: the same coding definition file is run
|
Settles the coding step: the same coding definition file is run
|
||||||
three times independently, and this command counts the votes and
|
three times independently, and this command counts the votes and
|
||||||
writes the final coding table the paper cites, as the CSV file
|
writes the final coding table the paper cites. A (song, keyword)
|
||||||
given as the fourth positional command-line argument. Only the
|
pair is written out when at least two of the three runs assign
|
||||||
keyword key sets of the three runs' archived ``output.jsonl``
|
it, carrying the pooled, deduplicated lyric quotes of the runs
|
||||||
files take part in the tally; the lyric quotes never do. A
|
that assigned it. ``--corrections`` names a CSV file of
|
||||||
(song, keyword) pair is written out when at least two of the
|
researcher-reviewed repairs to a run's records, applied before
|
||||||
three runs assign it, so three votes never tie, and it carries
|
the tally. ``--valid-keywords`` names a plain text file of the
|
||||||
the lyric quotes of every run that assigned it, pooled,
|
allowed keywords that every record's keywords must appear in.
|
||||||
deduplicated, sorted by Unicode code point, and joined with a
|
The songs are named from the working store, so this command runs
|
||||||
single ``|``: the three runs are peers, so the quote order
|
after ``build-db``. When any input is malformed, the tally fails
|
||||||
follows the text alone. A quote carries the lyric line-break
|
and nothing is written; the error message names what failed.
|
||||||
convention ``" / "`` where the lyric has a newline, applied once
|
|
||||||
when the run records load, so the corrections, the coding table,
|
|
||||||
and the database all share the one representation and nothing is
|
|
||||||
ever converted back. No lyric of the 883-song corpus contains
|
|
||||||
``" / "`` -- a corpus fact checked exhaustively, not a structural
|
|
||||||
guarantee -- so the convention is unambiguous here. The three
|
|
||||||
archives must cover exactly the same set of song IDs, every
|
|
||||||
record must be a successful result, and every record's "text"
|
|
||||||
must parse to a JSON object; otherwise the tally fails and
|
|
||||||
nothing is written.
|
|
||||||
|
|
||||||
Two optional inputs guard the tally. ``--corrections`` names a
|
|
||||||
CSV file of researcher-reviewed repairs, applied to each run's
|
|
||||||
records before anything else happens: a keyword row renames or
|
|
||||||
drops one keyword assignment of one song in one run, and an
|
|
||||||
evidence row rewrites or drops one lyric quote string wherever it
|
|
||||||
appears in that song's record for that run. Its two text fields
|
|
||||||
carry the two characters ``\n`` where the text has a newline, so
|
|
||||||
the file holds one row per line. Every row must match, so a
|
|
||||||
stale row fails the run. ``--valid-keywords`` names a
|
|
||||||
plain text file of the allowed keywords, one per line; once the
|
|
||||||
corrections are in, every keyword left in any record must appear
|
|
||||||
in it. The order is fixed and matters: the corrections come
|
|
||||||
first, so a repair may reunite the votes of a misspelled keyword
|
|
||||||
that the check would otherwise reject. With neither option, no
|
|
||||||
record is touched and no vocabulary is checked.
|
|
||||||
|
|
||||||
The archives identify a song as ``song-<ID>``, where ``<ID>`` is
|
|
||||||
the song's ID in the SQLite working store. The output table does
|
|
||||||
not carry that ID: every song is looked up in the working store
|
|
||||||
and written as its title and its stored artist credit instead, so
|
|
||||||
this command runs after ``build-db``. The step is fully
|
|
||||||
deterministic; no LLM call is made.
|
|
||||||
"""
|
"""
|
||||||
import argparse
|
import argparse
|
||||||
import csv
|
import csv
|
||||||
@@ -127,11 +94,6 @@ class Correction:
|
|||||||
class CorrectionTable:
|
class CorrectionTable:
|
||||||
"""The researcher-reviewed repairs of the runs' records."""
|
"""The researcher-reviewed repairs of the runs' records."""
|
||||||
|
|
||||||
MANUAL_CORRECTIONS_CSV: ClassVar[str] \
|
|
||||||
= "coding-corrections.csv"
|
|
||||||
"""The correction table CSV file's conventional name under
|
|
||||||
``data/manual/``."""
|
|
||||||
|
|
||||||
path: Path
|
path: Path
|
||||||
"""The correction table CSV file the repairs came from."""
|
"""The correction table CSV file the repairs came from."""
|
||||||
corrections: list[Correction]
|
corrections: list[Correction]
|
||||||
@@ -141,7 +103,7 @@ class CorrectionTable:
|
|||||||
class CorrectionsLoader:
|
class CorrectionsLoader:
|
||||||
"""The loader of the researcher-reviewed correction table."""
|
"""The loader of the researcher-reviewed correction table."""
|
||||||
|
|
||||||
__HEADER: tuple[str, str, str, str, str] = (
|
__HEADER: ClassVar[tuple[str, str, str, str, str]] = (
|
||||||
"Song ID", "Run", "Type", "To Be Replaced", "Correct Term")
|
"Song ID", "Run", "Type", "To Be Replaced", "Correct Term")
|
||||||
"""The header row the correction table CSV file must carry."""
|
"""The header row the correction table CSV file must carry."""
|
||||||
|
|
||||||
@@ -164,10 +126,8 @@ class CorrectionsLoader:
|
|||||||
of the runs the command was given, and a known type. The
|
of the runs the command was given, and a known type. The
|
||||||
file is read with the CSV reader, so a quoted field may
|
file is read with the CSV reader, so a quoted field may
|
||||||
hold a comma or a double quote. No field holds a line
|
hold a comma or a double quote. No field holds a line
|
||||||
break: the two text fields carry the lyric line-break
|
break (see the line-break convention on
|
||||||
convention ``" / "`` where the text has a newline -- the
|
``CodingTallier``). Nothing is written.
|
||||||
same representation the loaded run records carry -- and
|
|
||||||
are matched and applied verbatim. Nothing is written.
|
|
||||||
|
|
||||||
:return: The repairs, in file order.
|
:return: The repairs, in file order.
|
||||||
:raises TallyError: When the file cannot be read, the
|
:raises TallyError: When the file cannot be read, the
|
||||||
@@ -340,16 +300,16 @@ class TalliedCodings:
|
|||||||
class CodingTallier:
|
class CodingTallier:
|
||||||
"""The tallier of the three coding runs' keyword votes."""
|
"""The tallier of the three coding runs' keyword votes."""
|
||||||
|
|
||||||
__MAJORITY: int = 2
|
__MAJORITY: ClassVar[int] = 2
|
||||||
"""The number of runs that must assign a keyword to a song for
|
"""The number of runs that must assign a keyword to a song for
|
||||||
that code to be settled."""
|
that code to be settled."""
|
||||||
__MAX_REPORTED_IDS: int = 10
|
__MAX_REPORTED_IDS: ClassVar[int] = 10
|
||||||
"""The number of song IDs an error message lists before
|
"""The number of song IDs an error message lists before
|
||||||
summarizing the rest as a count."""
|
summarizing the rest as a count."""
|
||||||
__QUOTE_SEPARATOR: str = "|"
|
__QUOTE_SEPARATOR: ClassVar[str] = "|"
|
||||||
"""The separator between the distinct lyric quotes of one
|
"""The separator between the distinct lyric quotes of one
|
||||||
settled code."""
|
settled code."""
|
||||||
__LINE_BREAK: str = " / "
|
__LINE_BREAK: ClassVar[str] = " / "
|
||||||
"""The lyric line-break convention replacing every LF inside a
|
"""The lyric line-break convention replacing every LF inside a
|
||||||
quote. Unambiguous for this corpus only: none of the 883
|
quote. Unambiguous for this corpus only: none of the 883
|
||||||
songs' lyrics contains the three characters, checked
|
songs' lyrics contains the three characters, checked
|
||||||
@@ -616,11 +576,8 @@ class CodingTallier:
|
|||||||
if line.strip() == "":
|
if line.strip() == "":
|
||||||
continue
|
continue
|
||||||
record: Any = cls.__parse_json(line, str(path))
|
record: Any = cls.__parse_json(line, str(path))
|
||||||
if not isinstance(record, dict) or "id" not in record:
|
|
||||||
raise ValueError(
|
|
||||||
f"{path}: record without \"id\": {line}")
|
|
||||||
item_id: Any = record["id"]
|
item_id: Any = record["id"]
|
||||||
if "error" in record or "text" not in record:
|
if "text" not in record:
|
||||||
raise ValueError(
|
raise ValueError(
|
||||||
f"{path}: id {item_id}: not a successful"
|
f"{path}: id {item_id}: not a successful"
|
||||||
" result")
|
" result")
|
||||||
@@ -647,8 +604,8 @@ class CodingTallier:
|
|||||||
:param label: The location of the record, for the error
|
:param label: The location of the record, for the error
|
||||||
message.
|
message.
|
||||||
:return: The lyric quotes of every keyword, in the given
|
:return: The lyric quotes of every keyword, in the given
|
||||||
order, every LF inside a quote turned into the lyric
|
order, every LF inside a quote turned into the
|
||||||
line-break convention ``" / "``.
|
line-break convention.
|
||||||
:raises ValueError: When a keyword's value is not a list
|
:raises ValueError: When a keyword's value is not a list
|
||||||
of strings.
|
of strings.
|
||||||
"""
|
"""
|
||||||
@@ -739,7 +696,7 @@ class CodingTallier:
|
|||||||
first: set[int] = set(runs[0])
|
first: set[int] = set(runs[0])
|
||||||
index: int
|
index: int
|
||||||
records: dict[int, dict[str, list[str]]]
|
records: dict[int, dict[str, list[str]]]
|
||||||
for index, records in enumerate(runs):
|
for index, records in enumerate(runs[1:], start=1):
|
||||||
song_ids: set[int] = set(records)
|
song_ids: set[int] = set(records)
|
||||||
if song_ids == first:
|
if song_ids == first:
|
||||||
continue
|
continue
|
||||||
@@ -813,9 +770,6 @@ class CodingTallier:
|
|||||||
class CodingTable:
|
class CodingTable:
|
||||||
"""The final coding table the paper cites."""
|
"""The final coding table the paper cites."""
|
||||||
|
|
||||||
RESULT_CODINGS_CSV: ClassVar[str] = "codings.csv"
|
|
||||||
"""The coding table CSV file's conventional name under
|
|
||||||
``results/``."""
|
|
||||||
__HEADER: ClassVar[tuple[str, str, str, str]] \
|
__HEADER: ClassVar[tuple[str, str, str, str]] \
|
||||||
= ("Song", "Artist Credit", "Keyword", "Quote")
|
= ("Song", "Artist Credit", "Keyword", "Quote")
|
||||||
"""The header row of the coding table CSV file."""
|
"""The header row of the coding table CSV file."""
|
||||||
@@ -833,11 +787,8 @@ class CodingTable:
|
|||||||
endings, carrying the header row
|
endings, carrying the header row
|
||||||
``Song,Artist Credit,Keyword,Quote`` and one row per
|
``Song,Artist Credit,Keyword,Quote`` and one row per
|
||||||
settled keyword, in the row order. Every field is
|
settled keyword, in the row order. Every field is
|
||||||
written verbatim; a quote carries the lyric line-break
|
written verbatim, so the file holds one row per line.
|
||||||
convention ``" / "`` where the lyric has a line break, so
|
The parent directory is created when it does not exist.
|
||||||
no field holds a line break and the file holds one row
|
|
||||||
per line. The parent directory is created when it does
|
|
||||||
not exist.
|
|
||||||
|
|
||||||
:param output_csv: The output CSV file.
|
:param output_csv: The output CSV file.
|
||||||
:return: None.
|
:return: None.
|
||||||
@@ -970,8 +921,7 @@ def parse_args(argv: list[str] | None) -> argparse.Namespace:
|
|||||||
help="the third coding run's archive directory")
|
help="the third coding run's archive directory")
|
||||||
parser.add_argument(
|
parser.add_argument(
|
||||||
"output_csv", type=Path,
|
"output_csv", type=Path,
|
||||||
help="the output CSV file, by convention"
|
help="the output CSV file")
|
||||||
f" results/{CodingTable.RESULT_CODINGS_CSV}")
|
|
||||||
parser.add_argument(
|
parser.add_argument(
|
||||||
"--valid-keywords", type=Path, default=None,
|
"--valid-keywords", type=Path, default=None,
|
||||||
help="a plain text file of the allowed keywords, one per"
|
help="a plain text file of the allowed keywords, one per"
|
||||||
@@ -980,33 +930,23 @@ def parse_args(argv: list[str] | None) -> argparse.Namespace:
|
|||||||
parser.add_argument(
|
parser.add_argument(
|
||||||
"--corrections", type=Path, default=None,
|
"--corrections", type=Path, default=None,
|
||||||
help="the researcher-reviewed correction table CSV file,"
|
help="the researcher-reviewed correction table CSV file,"
|
||||||
" by convention"
|
|
||||||
f" data/manual/{CorrectionTable.MANUAL_CORRECTIONS_CSV},"
|
|
||||||
" applied to the runs' records before the tally"
|
" applied to the runs' records before the tally"
|
||||||
" (default: no repair)")
|
" (default: no repair)")
|
||||||
return parser.parse_args(argv)
|
return parser.parse_args(argv)
|
||||||
|
|
||||||
|
|
||||||
def main(argv: list[str] | None = None) -> int:
|
def main(argv: list[str] | None = None) -> int:
|
||||||
r"""Settle the coding by a majority of the three coding runs.
|
"""Settle the coding by a majority of the three coding runs.
|
||||||
|
|
||||||
Writes the final coding table as the given CSV file, holding
|
Writes the final coding table CSV file described in the
|
||||||
the header row ``Song,Artist Credit,Keyword,Quote`` and one
|
module docstring. The records are repaired from the
|
||||||
row per keyword at least two of the three runs assign, the
|
``--corrections`` table and then checked against the
|
||||||
song named by its title and its stored artist credit from the
|
``--valid-keywords`` list, when either is given. Nothing is
|
||||||
SQLite working store, and the keyword carrying the pooled,
|
written when the three archives do not cover the same songs,
|
||||||
deduplicated, and sorted lyric quotes of the runs that
|
a record is not a successful result, a correction is invalid
|
||||||
assigned it, joined with a single ``|`` and carrying the
|
or matches nothing, a keyword is not in the valid keyword
|
||||||
lyric line-break convention ``" / "`` where the lyric has a
|
list, or a song is not in the working store; the error message
|
||||||
newline, so the table holds one row per line. The records are
|
names what failed.
|
||||||
repaired from the ``--corrections`` table and then checked
|
|
||||||
against the ``--valid-keywords`` list, when either is given.
|
|
||||||
Nothing is written when the three archives do not cover the
|
|
||||||
same songs, a record is not a successful result, a record's
|
|
||||||
"text" does not parse to a JSON object of quote string lists,
|
|
||||||
a correction is invalid or matches nothing, a keyword is not
|
|
||||||
in the valid keyword list, or a song is not in the working
|
|
||||||
store; the error message names what failed.
|
|
||||||
|
|
||||||
:param argv: The command-line arguments, or None for
|
:param argv: The command-line arguments, or None for
|
||||||
``sys.argv``.
|
``sys.argv``.
|
||||||
|
|||||||
@@ -8,46 +8,281 @@
|
|||||||
Settles the semantic code groups of step 4: the same group
|
Settles the semantic code groups of step 4: the same group
|
||||||
selection definition file is run three times independently, and
|
selection definition file is run three times independently, and
|
||||||
this command counts the votes and writes the final group table
|
this command counts the votes and writes the final group table
|
||||||
the paper cites, as the CSV file given as the last positional
|
the paper cites. A (group, keyword) pair is written out when at
|
||||||
command-line argument. A (group, keyword) pair is written out
|
least two of the three runs select it. When an input is
|
||||||
when at least two of the three runs select it, so three votes
|
malformed, the tally fails and nothing is written; the error
|
||||||
never tie. A selected item that is not in the valid keyword
|
message names what failed.
|
||||||
list is invalid and casts no vote; every dropped occurrence is
|
|
||||||
reported on standard error. The group name is the record ID
|
|
||||||
with its ``group-`` prefix dropped. The rows are ordered by the
|
|
||||||
group name and then by the keyword, by Unicode code point, and
|
|
||||||
the file carries the header row ``Group,Keyword,Votes`` with
|
|
||||||
CRLF line endings per RFC 4180.
|
|
||||||
|
|
||||||
The three archives must cover exactly the same set of group IDs,
|
|
||||||
every record ID must carry the ``group-`` prefix, and every
|
|
||||||
record's "text" must parse to a JSON array of strings; otherwise
|
|
||||||
the tally fails and nothing is written.
|
|
||||||
"""
|
"""
|
||||||
import argparse
|
import argparse
|
||||||
import csv
|
import csv
|
||||||
import json
|
import json
|
||||||
import sys
|
import sys
|
||||||
import time
|
import time
|
||||||
|
from dataclasses import dataclass
|
||||||
from pathlib import Path
|
from pathlib import Path
|
||||||
from typing import Any
|
from typing import Any, ClassVar
|
||||||
|
|
||||||
from ..utils import format_duration
|
from ..utils import format_duration
|
||||||
|
|
||||||
GROUP_ID_PREFIX: str = "group-"
|
|
||||||
"""The prefix every group record ID must carry; the group name
|
|
||||||
is the rest of the ID."""
|
|
||||||
MAJORITY: int = 2
|
|
||||||
"""The number of runs that must select a keyword for a group for
|
|
||||||
that pair to be settled."""
|
|
||||||
HEADER: tuple[str, str, str] = ("Group", "Keyword", "Votes")
|
|
||||||
"""The header row of the group table CSV file."""
|
|
||||||
|
|
||||||
|
|
||||||
class TallyError(Exception):
|
class TallyError(Exception):
|
||||||
"""An error that fails the group tally."""
|
"""An error that fails the group tally."""
|
||||||
|
|
||||||
|
|
||||||
|
@dataclass(frozen=True)
|
||||||
|
class TalliedGroups:
|
||||||
|
"""The outcome of settling the group table."""
|
||||||
|
|
||||||
|
codes: int
|
||||||
|
"""The number of settled (group, keyword) pairs."""
|
||||||
|
groups: int
|
||||||
|
"""The number of groups the three runs cover."""
|
||||||
|
|
||||||
|
|
||||||
|
class GroupTallier:
|
||||||
|
"""The tallier of the three group-selection runs' votes."""
|
||||||
|
|
||||||
|
__GROUP_ID_PREFIX: ClassVar[str] = "group-"
|
||||||
|
"""The prefix every group record ID must carry; the group
|
||||||
|
name is the rest of the ID."""
|
||||||
|
__MAJORITY: ClassVar[int] = 2
|
||||||
|
"""The number of runs that must select a keyword for a group
|
||||||
|
for that pair to be settled."""
|
||||||
|
__HEADER: ClassVar[tuple[str, str, str]] \
|
||||||
|
= ("Group", "Keyword", "Votes")
|
||||||
|
"""The header row of the group table CSV file."""
|
||||||
|
|
||||||
|
def __init__(self, run_dir_1: Path, run_dir_2: Path,
|
||||||
|
run_dir_3: Path, valid_keywords_txt: Path,
|
||||||
|
output_csv: Path) -> None:
|
||||||
|
"""Set up the tallier of the three selection runs.
|
||||||
|
|
||||||
|
:param run_dir_1: The first selection run's archive
|
||||||
|
directory, containing ``output.jsonl``.
|
||||||
|
:param run_dir_2: The second selection run's archive
|
||||||
|
directory, containing ``output.jsonl``.
|
||||||
|
:param run_dir_3: The third selection run's archive
|
||||||
|
directory, containing ``output.jsonl``.
|
||||||
|
:param valid_keywords_txt: The plain text file of the
|
||||||
|
allowed keywords, one per line.
|
||||||
|
:param output_csv: The output group table CSV file.
|
||||||
|
"""
|
||||||
|
self.__run_dirs: list[Path] = [
|
||||||
|
run_dir_1, run_dir_2, run_dir_3]
|
||||||
|
"""The three runs' archive directories, in the given
|
||||||
|
order."""
|
||||||
|
self.__valid_keywords_txt: Path = valid_keywords_txt
|
||||||
|
"""The plain text file of the allowed keywords, one per
|
||||||
|
line."""
|
||||||
|
self.__output_csv: Path = output_csv
|
||||||
|
"""The output group table CSV file."""
|
||||||
|
|
||||||
|
def run(self) -> TalliedGroups:
|
||||||
|
"""Load the three runs, tally their votes, and write the
|
||||||
|
table.
|
||||||
|
|
||||||
|
:return: The settled code count and the group count.
|
||||||
|
:raises TallyError: When a file cannot be read, a line is
|
||||||
|
not a well-formed output record, a record is not a
|
||||||
|
successful result, an ID lacks the ``group-`` prefix,
|
||||||
|
a group has two records, a "text" does not parse to a
|
||||||
|
JSON array of strings, or the three runs do not cover
|
||||||
|
the same set of groups.
|
||||||
|
:raises OSError: When the output file cannot be written.
|
||||||
|
"""
|
||||||
|
valid: set[str] = self.__load_valid_keywords(
|
||||||
|
self.__valid_keywords_txt)
|
||||||
|
runs: list[dict[str, set[str]]] = [
|
||||||
|
self.__load_run(x) for x in self.__run_dirs]
|
||||||
|
self.__drop_invalid(runs, self.__run_dirs, valid)
|
||||||
|
rows: list[tuple[str, str, int]] = self.__tally(runs)
|
||||||
|
self.__write_csv(rows)
|
||||||
|
return TalliedGroups(codes=len(rows), groups=len(runs[0]))
|
||||||
|
|
||||||
|
@staticmethod
|
||||||
|
def __load_valid_keywords(path: Path) -> set[str]:
|
||||||
|
"""Load the valid keyword list.
|
||||||
|
|
||||||
|
:param path: The plain text file of the allowed keywords,
|
||||||
|
one per line.
|
||||||
|
:return: The allowed keywords.
|
||||||
|
:raises TallyError: When the file cannot be read or holds
|
||||||
|
no keyword.
|
||||||
|
"""
|
||||||
|
text: str
|
||||||
|
try:
|
||||||
|
text = path.read_text(encoding="utf-8")
|
||||||
|
except OSError as error:
|
||||||
|
raise TallyError(str(error)) from error
|
||||||
|
keywords: set[str] = {x.strip() for x in text.split("\n")
|
||||||
|
if x.strip() != ""}
|
||||||
|
if len(keywords) == 0:
|
||||||
|
raise TallyError(f"{path}: no keywords")
|
||||||
|
return keywords
|
||||||
|
|
||||||
|
@classmethod
|
||||||
|
def __load_run(cls, run_dir: Path) -> dict[str, set[str]]:
|
||||||
|
"""Load and validate the selection records of one run.
|
||||||
|
|
||||||
|
:param run_dir: The run's archive directory, containing
|
||||||
|
``output.jsonl``.
|
||||||
|
:return: The selected keywords of every group of the run,
|
||||||
|
keyed by the group name, the duplicates within one
|
||||||
|
record's selection collapsed.
|
||||||
|
:raises TallyError: When the file cannot be read, a line
|
||||||
|
is not a well-formed output record, a record is not a
|
||||||
|
successful result, an ID lacks the ``group-`` prefix,
|
||||||
|
a group has two records, or a "text" does not parse
|
||||||
|
to a JSON array of strings.
|
||||||
|
"""
|
||||||
|
path: Path = run_dir / "output.jsonl"
|
||||||
|
text: str
|
||||||
|
try:
|
||||||
|
text = path.read_text(encoding="utf-8")
|
||||||
|
except OSError as error:
|
||||||
|
raise TallyError(str(error)) from error
|
||||||
|
records: dict[str, set[str]] = {}
|
||||||
|
line: str
|
||||||
|
for line in text.split("\n"):
|
||||||
|
if line.strip() == "":
|
||||||
|
continue
|
||||||
|
record: Any
|
||||||
|
try:
|
||||||
|
record = json.loads(line)
|
||||||
|
except json.JSONDecodeError as error:
|
||||||
|
raise TallyError(
|
||||||
|
f"{path}: malformed JSON: {error}") from error
|
||||||
|
item_id: Any = record["id"]
|
||||||
|
if "text" not in record:
|
||||||
|
raise TallyError(
|
||||||
|
f"{path}: id {item_id}: not a successful"
|
||||||
|
" result")
|
||||||
|
if not isinstance(item_id, str) \
|
||||||
|
or not item_id.startswith(
|
||||||
|
cls.__GROUP_ID_PREFIX) \
|
||||||
|
or item_id == cls.__GROUP_ID_PREFIX:
|
||||||
|
raise TallyError(
|
||||||
|
f"{path}: id {item_id}: not in the"
|
||||||
|
f" \"{cls.__GROUP_ID_PREFIX}<name>\" form")
|
||||||
|
group: str = item_id[len(cls.__GROUP_ID_PREFIX):]
|
||||||
|
if group in records:
|
||||||
|
raise TallyError(
|
||||||
|
f"{path}: id {item_id}: duplicate record")
|
||||||
|
records[group] = cls.__parse_selection(
|
||||||
|
record["text"], f"{path}: id {item_id}")
|
||||||
|
return records
|
||||||
|
|
||||||
|
@staticmethod
|
||||||
|
def __parse_selection(text: Any, label: str) -> set[str]:
|
||||||
|
"""Parse and validate the selected keywords of one record.
|
||||||
|
|
||||||
|
:param text: The "text" field of the record.
|
||||||
|
:param label: The location of the record, for the error
|
||||||
|
message.
|
||||||
|
:return: The selected keywords, the duplicates collapsed.
|
||||||
|
:raises TallyError: When the text does not parse to a
|
||||||
|
JSON array of strings.
|
||||||
|
"""
|
||||||
|
selected: Any
|
||||||
|
try:
|
||||||
|
selected = json.loads(text)
|
||||||
|
except json.JSONDecodeError as error:
|
||||||
|
raise TallyError(
|
||||||
|
f"{label}: \"text\" is malformed JSON:"
|
||||||
|
f" {error}") from error
|
||||||
|
if not isinstance(selected, list) \
|
||||||
|
or not all(isinstance(x, str) for x in selected):
|
||||||
|
raise TallyError(
|
||||||
|
f"{label}: \"text\" does not parse to a JSON"
|
||||||
|
" array of strings")
|
||||||
|
return set(selected)
|
||||||
|
|
||||||
|
@staticmethod
|
||||||
|
def __drop_invalid(runs: list[dict[str, set[str]]],
|
||||||
|
run_dirs: list[Path],
|
||||||
|
valid: set[str]) -> None:
|
||||||
|
"""Drop the out-of-vocabulary selections of every run.
|
||||||
|
|
||||||
|
Every dropped occurrence is reported on standard error as
|
||||||
|
an observable side effect.
|
||||||
|
|
||||||
|
:param runs: The runs' records, filtered in place.
|
||||||
|
:param run_dirs: The run directories, for the messages.
|
||||||
|
:param valid: The allowed keywords.
|
||||||
|
:return: None.
|
||||||
|
"""
|
||||||
|
records: dict[str, set[str]]
|
||||||
|
run_dir: Path
|
||||||
|
for records, run_dir in zip(runs, run_dirs):
|
||||||
|
group: str
|
||||||
|
selected: set[str]
|
||||||
|
for group, selected in records.items():
|
||||||
|
keyword: str
|
||||||
|
for keyword in sorted(selected - valid):
|
||||||
|
print(
|
||||||
|
f"note: {run_dir.name} group-{group}:"
|
||||||
|
f" dropped out-of-vocabulary item"
|
||||||
|
f" \"{keyword}\"", file=sys.stderr)
|
||||||
|
records[group] = selected & valid
|
||||||
|
|
||||||
|
@classmethod
|
||||||
|
def __tally(cls, runs: list[dict[str, set[str]]]) \
|
||||||
|
-> list[tuple[str, str, int]]:
|
||||||
|
"""Tally the keyword votes of the runs, group by group.
|
||||||
|
|
||||||
|
:param runs: The runs' records, all covering the same set
|
||||||
|
of groups.
|
||||||
|
:return: The settled rows, each the group name, the
|
||||||
|
keyword, and the number of votes, ordered by the
|
||||||
|
group name and then by the keyword, by Unicode code
|
||||||
|
point.
|
||||||
|
:raises TallyError: When the runs do not cover the same
|
||||||
|
set of groups.
|
||||||
|
"""
|
||||||
|
groups: set[str] = set(runs[0])
|
||||||
|
records: dict[str, set[str]]
|
||||||
|
for records in runs[1:]:
|
||||||
|
if set(records) != groups:
|
||||||
|
raise TallyError(
|
||||||
|
"the three runs do not cover the same"
|
||||||
|
" groups: " + ", ".join(sorted(
|
||||||
|
groups.symmetric_difference(
|
||||||
|
set(records)))))
|
||||||
|
rows: list[tuple[str, str, int]] = []
|
||||||
|
group: str
|
||||||
|
for group in sorted(groups):
|
||||||
|
votes: dict[str, int] = {}
|
||||||
|
for records in runs:
|
||||||
|
keyword: str
|
||||||
|
for keyword in records[group]:
|
||||||
|
votes[keyword] = votes.get(keyword, 0) + 1
|
||||||
|
rows.extend(
|
||||||
|
(group, x, votes[x])
|
||||||
|
for x in sorted(votes) if votes[x] >= cls.__MAJORITY)
|
||||||
|
return rows
|
||||||
|
|
||||||
|
def __write_csv(self, rows: list[tuple[str, str, int]]) \
|
||||||
|
-> None:
|
||||||
|
"""Write the group table CSV file.
|
||||||
|
|
||||||
|
Writes an RFC 4180 CSV file, UTF-8, with CRLF line
|
||||||
|
endings, carrying the header row ``Group,Keyword,Votes``
|
||||||
|
and one row per settled (group, keyword) pair, in the row
|
||||||
|
order. The parent directory is created when it does not
|
||||||
|
exist.
|
||||||
|
|
||||||
|
:param rows: The settled rows, in the output order.
|
||||||
|
:return: None.
|
||||||
|
:raises OSError: When the file cannot be written.
|
||||||
|
"""
|
||||||
|
self.__output_csv.parent.mkdir(parents=True, exist_ok=True)
|
||||||
|
with open(self.__output_csv, "w", encoding="utf-8",
|
||||||
|
newline="") as file:
|
||||||
|
writer: Any = csv.writer(file)
|
||||||
|
writer.writerow(self.__HEADER)
|
||||||
|
writer.writerows(rows)
|
||||||
|
|
||||||
|
|
||||||
def parse_args(argv: list[str] | None) -> argparse.Namespace:
|
def parse_args(argv: list[str] | None) -> argparse.Namespace:
|
||||||
"""Parse the command-line arguments.
|
"""Parse the command-line arguments.
|
||||||
|
|
||||||
@@ -68,238 +303,34 @@ def parse_args(argv: list[str] | None) -> argparse.Namespace:
|
|||||||
"run_dir_3", type=Path,
|
"run_dir_3", type=Path,
|
||||||
help="the third selection run's archive directory")
|
help="the third selection run's archive directory")
|
||||||
parser.add_argument(
|
parser.add_argument(
|
||||||
"valid_keywords", type=Path,
|
"output_csv", type=Path,
|
||||||
|
help="the output CSV file")
|
||||||
|
parser.add_argument(
|
||||||
|
"--valid-keywords", type=Path, required=True,
|
||||||
help="a plain text file of the allowed keywords, one per"
|
help="a plain text file of the allowed keywords, one per"
|
||||||
" line")
|
" line")
|
||||||
parser.add_argument(
|
|
||||||
"output_csv", type=Path,
|
|
||||||
help="the output CSV file, by convention"
|
|
||||||
" results/groups.csv")
|
|
||||||
return parser.parse_args(argv)
|
return parser.parse_args(argv)
|
||||||
|
|
||||||
|
|
||||||
def load_valid_keywords(path: Path) -> set[str]:
|
|
||||||
"""Load the valid keyword list.
|
|
||||||
|
|
||||||
:param path: The plain text file of the allowed keywords, one
|
|
||||||
per line.
|
|
||||||
:return: The allowed keywords.
|
|
||||||
:raises TallyError: When the file cannot be read or holds no
|
|
||||||
keyword.
|
|
||||||
"""
|
|
||||||
text: str
|
|
||||||
try:
|
|
||||||
text = path.read_text(encoding="utf-8")
|
|
||||||
except OSError as error:
|
|
||||||
raise TallyError(str(error)) from error
|
|
||||||
keywords: set[str] = {x.strip() for x in text.split("\n")
|
|
||||||
if x.strip() != ""}
|
|
||||||
if len(keywords) == 0:
|
|
||||||
raise TallyError(f"{path}: no keywords")
|
|
||||||
return keywords
|
|
||||||
|
|
||||||
|
|
||||||
def load_run(run_dir: Path) -> dict[str, set[str]]:
|
|
||||||
"""Load and validate the selection records of one run.
|
|
||||||
|
|
||||||
:param run_dir: The run's archive directory, containing
|
|
||||||
``output.jsonl``.
|
|
||||||
:return: The selected keywords of every group of the run,
|
|
||||||
keyed by the group name, the duplicates within one
|
|
||||||
record's selection collapsed.
|
|
||||||
:raises TallyError: When the file cannot be read, a line is
|
|
||||||
not a well-formed output record, a record is not a
|
|
||||||
successful result, an ID lacks the ``group-`` prefix, a
|
|
||||||
group has two records, or a "text" does not parse to a
|
|
||||||
JSON array of strings.
|
|
||||||
"""
|
|
||||||
path: Path = run_dir / "output.jsonl"
|
|
||||||
text: str
|
|
||||||
try:
|
|
||||||
text = path.read_text(encoding="utf-8")
|
|
||||||
except OSError as error:
|
|
||||||
raise TallyError(str(error)) from error
|
|
||||||
records: dict[str, set[str]] = {}
|
|
||||||
line: str
|
|
||||||
for line in text.split("\n"):
|
|
||||||
if line.strip() == "":
|
|
||||||
continue
|
|
||||||
record: Any
|
|
||||||
try:
|
|
||||||
record = json.loads(line)
|
|
||||||
except json.JSONDecodeError as error:
|
|
||||||
raise TallyError(
|
|
||||||
f"{path}: malformed JSON: {error}") from error
|
|
||||||
if not isinstance(record, dict) or "id" not in record:
|
|
||||||
raise TallyError(
|
|
||||||
f"{path}: record without \"id\": {line}")
|
|
||||||
item_id: Any = record["id"]
|
|
||||||
if "error" in record or "text" not in record:
|
|
||||||
raise TallyError(
|
|
||||||
f"{path}: id {item_id}: not a successful result")
|
|
||||||
if not isinstance(item_id, str) \
|
|
||||||
or not item_id.startswith(GROUP_ID_PREFIX) \
|
|
||||||
or item_id == GROUP_ID_PREFIX:
|
|
||||||
raise TallyError(
|
|
||||||
f"{path}: id {item_id}: not in the"
|
|
||||||
f" \"{GROUP_ID_PREFIX}<name>\" form")
|
|
||||||
group: str = item_id[len(GROUP_ID_PREFIX):]
|
|
||||||
if group in records:
|
|
||||||
raise TallyError(
|
|
||||||
f"{path}: id {item_id}: duplicate record")
|
|
||||||
records[group] = _selection(
|
|
||||||
record["text"], f"{path}: id {item_id}")
|
|
||||||
if len(records) == 0:
|
|
||||||
raise TallyError(f"{path}: no records")
|
|
||||||
return records
|
|
||||||
|
|
||||||
|
|
||||||
def _selection(text: Any, label: str) -> set[str]:
|
|
||||||
"""Parse and validate the selected keywords of one record.
|
|
||||||
|
|
||||||
:param text: The "text" field of the record.
|
|
||||||
:param label: The location of the record, for the error
|
|
||||||
message.
|
|
||||||
:return: The selected keywords, the duplicates collapsed.
|
|
||||||
:raises TallyError: When the text does not parse to a JSON
|
|
||||||
array of strings.
|
|
||||||
"""
|
|
||||||
if not isinstance(text, str):
|
|
||||||
raise TallyError(f"{label}: \"text\" is not a string")
|
|
||||||
selected: Any
|
|
||||||
try:
|
|
||||||
selected = json.loads(text)
|
|
||||||
except json.JSONDecodeError as error:
|
|
||||||
raise TallyError(
|
|
||||||
f"{label}: \"text\" is malformed JSON:"
|
|
||||||
f" {error}") from error
|
|
||||||
if not isinstance(selected, list) \
|
|
||||||
or not all(isinstance(x, str) for x in selected):
|
|
||||||
raise TallyError(
|
|
||||||
f"{label}: \"text\" does not parse to a JSON array"
|
|
||||||
" of strings")
|
|
||||||
return set(selected)
|
|
||||||
|
|
||||||
|
|
||||||
def drop_invalid(runs: list[dict[str, set[str]]],
|
|
||||||
run_dirs: list[Path],
|
|
||||||
valid: set[str]) -> None:
|
|
||||||
"""Drop the out-of-vocabulary selections of every run.
|
|
||||||
|
|
||||||
Every dropped occurrence is reported on standard error as an
|
|
||||||
observable side effect.
|
|
||||||
|
|
||||||
:param runs: The runs' records, filtered in place.
|
|
||||||
:param run_dirs: The run directories, for the messages.
|
|
||||||
:param valid: The allowed keywords.
|
|
||||||
:return: None.
|
|
||||||
"""
|
|
||||||
records: dict[str, set[str]]
|
|
||||||
run_dir: Path
|
|
||||||
for records, run_dir in zip(runs, run_dirs):
|
|
||||||
group: str
|
|
||||||
selected: set[str]
|
|
||||||
for group, selected in records.items():
|
|
||||||
keyword: str
|
|
||||||
for keyword in sorted(selected - valid):
|
|
||||||
print(
|
|
||||||
f"note: {run_dir.name} group-{group}:"
|
|
||||||
f" dropped out-of-vocabulary item"
|
|
||||||
f" \"{keyword}\"", file=sys.stderr)
|
|
||||||
records[group] = selected & valid
|
|
||||||
|
|
||||||
|
|
||||||
def tally(runs: list[dict[str, set[str]]]) \
|
|
||||||
-> list[tuple[str, str, int]]:
|
|
||||||
"""Tally the keyword votes of the runs, group by group.
|
|
||||||
|
|
||||||
:param runs: The runs' records, all covering the same set of
|
|
||||||
groups.
|
|
||||||
:return: The settled rows, each the group name, the keyword,
|
|
||||||
and the number of votes, ordered by the group name and
|
|
||||||
then by the keyword, by Unicode code point.
|
|
||||||
:raises TallyError: When the runs do not cover the same set
|
|
||||||
of groups.
|
|
||||||
"""
|
|
||||||
groups: set[str] = set(runs[0])
|
|
||||||
records: dict[str, set[str]]
|
|
||||||
for records in runs[1:]:
|
|
||||||
if set(records) != groups:
|
|
||||||
raise TallyError(
|
|
||||||
"the three runs do not cover the same groups: "
|
|
||||||
+ ", ".join(sorted(
|
|
||||||
groups.symmetric_difference(set(records)))))
|
|
||||||
rows: list[tuple[str, str, int]] = []
|
|
||||||
group: str
|
|
||||||
for group in sorted(groups):
|
|
||||||
votes: dict[str, int] = {}
|
|
||||||
for records in runs:
|
|
||||||
keyword: str
|
|
||||||
for keyword in records[group]:
|
|
||||||
votes[keyword] = votes.get(keyword, 0) + 1
|
|
||||||
rows.extend(
|
|
||||||
(group, x, votes[x])
|
|
||||||
for x in sorted(votes) if votes[x] >= MAJORITY)
|
|
||||||
return rows
|
|
||||||
|
|
||||||
|
|
||||||
def write_csv(output_csv: Path,
|
|
||||||
rows: list[tuple[str, str, int]]) -> None:
|
|
||||||
"""Write the group table CSV file.
|
|
||||||
|
|
||||||
Writes an RFC 4180 CSV file, UTF-8, with CRLF line endings,
|
|
||||||
carrying the header row ``Group,Keyword,Votes`` and one row
|
|
||||||
per settled (group, keyword) pair, in the row order. The
|
|
||||||
parent directory is created when it does not exist.
|
|
||||||
|
|
||||||
:param output_csv: The output CSV file.
|
|
||||||
:param rows: The settled rows, in the output order.
|
|
||||||
:return: None.
|
|
||||||
:raises OSError: When the file cannot be written.
|
|
||||||
"""
|
|
||||||
output_csv.parent.mkdir(parents=True, exist_ok=True)
|
|
||||||
with open(output_csv, "w", encoding="utf-8",
|
|
||||||
newline="") as file:
|
|
||||||
writer: Any = csv.writer(file)
|
|
||||||
writer.writerow(HEADER)
|
|
||||||
writer.writerows(rows)
|
|
||||||
|
|
||||||
|
|
||||||
def main(argv: list[str] | None = None) -> int:
|
def main(argv: list[str] | None = None) -> int:
|
||||||
"""Settle the code groups by a majority of the three runs.
|
"""Settle the code groups by a majority of the three runs.
|
||||||
|
|
||||||
Writes the final group table as the given CSV file, holding
|
|
||||||
the header row ``Group,Keyword,Votes`` and one row per
|
|
||||||
(group, keyword) pair at least two of the three runs select,
|
|
||||||
ordered by the group name and then by the keyword. A
|
|
||||||
selected item that is not in the valid keyword list casts no
|
|
||||||
vote, each dropped occurrence reported on standard error.
|
|
||||||
Nothing is written when the three archives do not cover the
|
|
||||||
same groups, a record is not a successful result, an ID lacks
|
|
||||||
the ``group-`` prefix, or a record's "text" does not parse to
|
|
||||||
a JSON array of strings; the error message names what failed.
|
|
||||||
|
|
||||||
:param argv: The command-line arguments, or None for
|
:param argv: The command-line arguments, or None for
|
||||||
``sys.argv``.
|
``sys.argv``.
|
||||||
:return: The exit status: 0 on success, non-zero on failure.
|
:return: The exit status: 0 on success, non-zero on failure.
|
||||||
"""
|
"""
|
||||||
started: float = time.monotonic()
|
started: float = time.monotonic()
|
||||||
args: argparse.Namespace = parse_args(argv)
|
args: argparse.Namespace = parse_args(argv)
|
||||||
run_dirs: list[Path] = [
|
|
||||||
args.run_dir_1, args.run_dir_2, args.run_dir_3]
|
|
||||||
try:
|
try:
|
||||||
valid: set[str] = load_valid_keywords(args.valid_keywords)
|
tallied: TalliedGroups = GroupTallier(
|
||||||
runs: list[dict[str, set[str]]] = [
|
args.run_dir_1, args.run_dir_2, args.run_dir_3,
|
||||||
load_run(x) for x in run_dirs]
|
args.valid_keywords, args.output_csv).run()
|
||||||
drop_invalid(runs, run_dirs, valid)
|
|
||||||
rows: list[tuple[str, str, int]] = tally(runs)
|
|
||||||
write_csv(args.output_csv, rows)
|
|
||||||
except (TallyError, OSError) as error:
|
except (TallyError, OSError) as error:
|
||||||
print(f"error: {error}", file=sys.stderr)
|
print(f"error: {error}", file=sys.stderr)
|
||||||
return 1
|
return 1
|
||||||
elapsed: str = format_duration(time.monotonic() - started)
|
elapsed: str = format_duration(time.monotonic() - started)
|
||||||
print(
|
print(
|
||||||
f"Done. Settled {len(rows)} codes across"
|
f"Done. Settled {tallied.codes} codes across"
|
||||||
f" {len(runs[0])} groups. {elapsed} elapsed.",
|
f" {tallied.groups} groups. {elapsed} elapsed.",
|
||||||
file=sys.stderr)
|
file=sys.stderr)
|
||||||
return 0
|
return 0
|
||||||
|
|||||||
@@ -285,9 +285,11 @@ class TestBuildDB(unittest.TestCase):
|
|||||||
patchers: list[Any] = [
|
patchers: list[Any] = [
|
||||||
mock.patch.object(build_db, "ds", self.__ds),
|
mock.patch.object(build_db, "ds", self.__ds),
|
||||||
mock.patch.object(
|
mock.patch.object(
|
||||||
build_db.SongImporter, "YEARS", [2016, 2017]),
|
build_db.SongImporter, "_SongImporter__YEARS",
|
||||||
|
[2016, 2017]),
|
||||||
mock.patch.object(
|
mock.patch.object(
|
||||||
build_db.SongImporter, "RANKS_PER_YEAR", 2)]
|
build_db.SongImporter,
|
||||||
|
"_SongImporter__RANKS_PER_YEAR", 2)]
|
||||||
for patcher in patchers:
|
for patcher in patchers:
|
||||||
patcher.start()
|
patcher.start()
|
||||||
self.addCleanup(patcher.stop)
|
self.addCleanup(patcher.stop)
|
||||||
@@ -1014,12 +1016,9 @@ class TestBuildDB(unittest.TestCase):
|
|||||||
"""Test that the coding CSV imports one row per song and
|
"""Test that the coding CSV imports one row per song and
|
||||||
keyword, storing the quote column verbatim."""
|
keyword, storing the quote column verbatim."""
|
||||||
self.__write_codings(self.CODINGS_CSV)
|
self.__write_codings(self.CODINGS_CSV)
|
||||||
status: int
|
self.assertEqual(
|
||||||
stderr: str
|
self.__run_build("--codings", str(self.__codings))[0],
|
||||||
status, stderr = self.__run_build(
|
0)
|
||||||
"--codings", str(self.__codings))
|
|
||||||
self.assertEqual(status, 0)
|
|
||||||
self.assertIn("3 codings", stderr)
|
|
||||||
self.assertEqual(
|
self.assertEqual(
|
||||||
self.__stored_codings(),
|
self.__stored_codings(),
|
||||||
{("Hello", "longing"):
|
{("Hello", "longing"):
|
||||||
@@ -1043,11 +1042,7 @@ class TestBuildDB(unittest.TestCase):
|
|||||||
def test_omitted_codings_leaves_table_empty(self) -> None:
|
def test_omitted_codings_leaves_table_empty(self) -> None:
|
||||||
"""Test that an omitted coding option leaves no codings."""
|
"""Test that an omitted coding option leaves no codings."""
|
||||||
self.__write_codings(self.CODINGS_CSV)
|
self.__write_codings(self.CODINGS_CSV)
|
||||||
status: int
|
self.assertEqual(self.__run_build()[0], 0)
|
||||||
stderr: str
|
|
||||||
status, stderr = self.__run_build()
|
|
||||||
self.assertEqual(status, 0)
|
|
||||||
self.assertIn("0 codings", stderr)
|
|
||||||
self.assertEqual(self.__stored_codings(), {})
|
self.assertEqual(self.__stored_codings(), {})
|
||||||
|
|
||||||
def test_codings_unknown_song_fails(self) -> None:
|
def test_codings_unknown_song_fails(self) -> None:
|
||||||
@@ -1119,12 +1114,9 @@ class TestBuildDB(unittest.TestCase):
|
|||||||
"Song,Artist Credit,Keyword,Quote\n"
|
"Song,Artist Credit,Keyword,Quote\n"
|
||||||
"Shape of You,Ed Sheeran,attraction,I'm in love with"
|
"Shape of You,Ed Sheeran,attraction,I'm in love with"
|
||||||
" your body\n")
|
" your body\n")
|
||||||
status: int
|
self.assertEqual(
|
||||||
stderr: str
|
self.__run_build("--codings", str(self.__codings))[0],
|
||||||
status, stderr = self.__run_build(
|
0)
|
||||||
"--codings", str(self.__codings))
|
|
||||||
self.assertEqual(status, 0)
|
|
||||||
self.assertIn("1 codings", stderr)
|
|
||||||
self.assertEqual(
|
self.assertEqual(
|
||||||
self.__stored_codings(),
|
self.__stored_codings(),
|
||||||
{("Shape of You", "attraction"):
|
{("Shape of You", "attraction"):
|
||||||
@@ -1159,12 +1151,8 @@ class TestBuildDB(unittest.TestCase):
|
|||||||
"""Test that the group CSV imports one row per group and
|
"""Test that the group CSV imports one row per group and
|
||||||
keyword, the votes stored as integers."""
|
keyword, the votes stored as integers."""
|
||||||
self.__write_groups(self.GROUPS_CSV)
|
self.__write_groups(self.GROUPS_CSV)
|
||||||
status: int
|
self.assertEqual(
|
||||||
stderr: str
|
self.__run_build("--groups", str(self.__groups))[0], 0)
|
||||||
status, stderr = self.__run_build(
|
|
||||||
"--groups", str(self.__groups))
|
|
||||||
self.assertEqual(status, 0)
|
|
||||||
self.assertIn("3 group members", stderr)
|
|
||||||
self.assertEqual(
|
self.assertEqual(
|
||||||
self.__stored_groups(),
|
self.__stored_groups(),
|
||||||
{("masculine", "dominance-and-power"): 3,
|
{("masculine", "dominance-and-power"): 3,
|
||||||
@@ -1180,12 +1168,8 @@ class TestBuildDB(unittest.TestCase):
|
|||||||
self.__write_groups(
|
self.__write_groups(
|
||||||
"Group,Keyword,Votes\n"
|
"Group,Keyword,Votes\n"
|
||||||
"vulnerable,longing-and-loss,3\n")
|
"vulnerable,longing-and-loss,3\n")
|
||||||
status: int
|
self.assertEqual(
|
||||||
stderr: str
|
self.__run_build("--groups", str(self.__groups))[0], 0)
|
||||||
status, stderr = self.__run_build(
|
|
||||||
"--groups", str(self.__groups))
|
|
||||||
self.assertEqual(status, 0)
|
|
||||||
self.assertIn("1 group members", stderr)
|
|
||||||
self.assertEqual(
|
self.assertEqual(
|
||||||
self.__stored_groups(),
|
self.__stored_groups(),
|
||||||
{("vulnerable", "longing-and-loss"): 3})
|
{("vulnerable", "longing-and-loss"): 3})
|
||||||
@@ -1407,14 +1391,11 @@ class TestBuildDB(unittest.TestCase):
|
|||||||
together, reflected in the final counts message."""
|
together, reflected in the final counts message."""
|
||||||
self.__write_patterns(self.PATTERNS_CSV)
|
self.__write_patterns(self.PATTERNS_CSV)
|
||||||
self.__write_annotations(self.ANNOTATIONS_CSV)
|
self.__write_annotations(self.ANNOTATIONS_CSV)
|
||||||
status: int
|
self.assertEqual(
|
||||||
stderr: str
|
self.__run_build(
|
||||||
status, stderr = self.__run_build(
|
"--patterns", str(self.__patterns),
|
||||||
"--patterns", str(self.__patterns),
|
"--annotations", str(self.__annotations))[0],
|
||||||
"--annotations", str(self.__annotations))
|
0)
|
||||||
self.assertEqual(status, 0)
|
|
||||||
self.assertIn("2 patterns", stderr)
|
|
||||||
self.assertIn("2 annotations", stderr)
|
|
||||||
self.assertEqual(
|
self.assertEqual(
|
||||||
self.__stored_patterns(),
|
self.__stored_patterns(),
|
||||||
{"M1": ("male", "Dominance",
|
{"M1": ("male", "Dominance",
|
||||||
@@ -1432,12 +1413,7 @@ class TestBuildDB(unittest.TestCase):
|
|||||||
annotation tables empty."""
|
annotation tables empty."""
|
||||||
self.__write_patterns(self.PATTERNS_CSV)
|
self.__write_patterns(self.PATTERNS_CSV)
|
||||||
self.__write_annotations(self.ANNOTATIONS_CSV)
|
self.__write_annotations(self.ANNOTATIONS_CSV)
|
||||||
status: int
|
self.assertEqual(self.__run_build()[0], 0)
|
||||||
stderr: str
|
|
||||||
status, stderr = self.__run_build()
|
|
||||||
self.assertEqual(status, 0)
|
|
||||||
self.assertIn("0 patterns", stderr)
|
|
||||||
self.assertIn("0 annotations", stderr)
|
|
||||||
self.assertEqual(self.__stored_patterns(), {})
|
self.assertEqual(self.__stored_patterns(), {})
|
||||||
self.assertEqual(self.__stored_annotations(), {})
|
self.assertEqual(self.__stored_annotations(), {})
|
||||||
|
|
||||||
|
|||||||
@@ -51,10 +51,12 @@ class TestFetchArtists(unittest.TestCase):
|
|||||||
self.addCleanup(self.__ds.engine.dispose)
|
self.addCleanup(self.__ds.engine.dispose)
|
||||||
patchers: list[Any] = [
|
patchers: list[Any] = [
|
||||||
mock.patch.object(fetch_artists, "ds", self.__ds),
|
mock.patch.object(fetch_artists, "ds", self.__ds),
|
||||||
mock.patch.object(fetch_artists, "SLEEP_SECONDS",
|
mock.patch.object(
|
||||||
0.0),
|
fetch_artists.ArtistFetcher,
|
||||||
mock.patch.object(fetch_artists, "RETRY_SECONDS",
|
"_ArtistFetcher__SLEEP_SECONDS", 0.0),
|
||||||
0.0)]
|
mock.patch.object(
|
||||||
|
fetch_artists.ArtistFetcher,
|
||||||
|
"_ArtistFetcher__RETRY_SECONDS", 0.0)]
|
||||||
for patcher in patchers:
|
for patcher in patchers:
|
||||||
patcher.start()
|
patcher.start()
|
||||||
self.addCleanup(patcher.stop)
|
self.addCleanup(patcher.stop)
|
||||||
@@ -315,10 +317,12 @@ class TestFetchArtists(unittest.TestCase):
|
|||||||
self.assertEqual(status, 0)
|
self.assertEqual(status, 0)
|
||||||
self.assertEqual(urlopen.call_count, 3)
|
self.assertEqual(urlopen.call_count, 3)
|
||||||
first: Any = urlopen.call_args_list[0][0][0]
|
first: Any = urlopen.call_args_list[0][0][0]
|
||||||
self.assertEqual(first.get_header("User-agent"),
|
self.assertEqual(
|
||||||
fetch_artists.USER_AGENT)
|
first.get_header("User-agent"),
|
||||||
self.assertTrue(
|
fetch_artists.ArtistFetcher._ArtistFetcher__USER_AGENT)
|
||||||
first.full_url.startswith(fetch_artists.SPARQL_URL))
|
self.assertTrue(first.full_url.startswith(
|
||||||
|
fetch_artists.ArtistFetcher
|
||||||
|
._ArtistFetcher__SPARQL_URL))
|
||||||
self.assertEqual(
|
self.assertEqual(
|
||||||
first.get_header("Accept"),
|
first.get_header("Accept"),
|
||||||
"application/sparql-results+json")
|
"application/sparql-results+json")
|
||||||
@@ -358,7 +362,8 @@ class TestFetchArtists(unittest.TestCase):
|
|||||||
qid, {}, "a brand the type gate excludes")
|
qid, {}, "a brand the type gate excludes")
|
||||||
urlopen: mock.Mock
|
urlopen: mock.Mock
|
||||||
with mock.patch.object(
|
with mock.patch.object(
|
||||||
fetch_artists, "PINNED_QIDS",
|
fetch_artists.ArtistFetcher,
|
||||||
|
"_ArtistFetcher__PINNED_QIDS",
|
||||||
{"Brandy Brand": qid}), \
|
{"Brandy Brand": qid}), \
|
||||||
mock.patch(
|
mock.patch(
|
||||||
"urllib.request.urlopen",
|
"urllib.request.urlopen",
|
||||||
@@ -369,7 +374,7 @@ class TestFetchArtists(unittest.TestCase):
|
|||||||
self.assertEqual(urlopen.call_count, 1)
|
self.assertEqual(urlopen.call_count, 1)
|
||||||
first: Any = urlopen.call_args_list[0][0][0]
|
first: Any = urlopen.call_args_list[0][0][0]
|
||||||
self.assertTrue(first.full_url.startswith(
|
self.assertTrue(first.full_url.startswith(
|
||||||
fetch_artists.API_URL))
|
fetch_artists.ArtistFetcher._ArtistFetcher__API_URL))
|
||||||
rows: list[list[str]] = self.__read_rows(
|
rows: list[list[str]] = self.__read_rows(
|
||||||
self.__snapshot)
|
self.__snapshot)
|
||||||
self.assertEqual(rows[1], [
|
self.assertEqual(rows[1], [
|
||||||
@@ -886,7 +891,8 @@ class TestFetchArtists(unittest.TestCase):
|
|||||||
"urllib.request.urlopen",
|
"urllib.request.urlopen",
|
||||||
side_effect=[self.__response(empty)]) as urlopen,
|
side_effect=[self.__response(empty)]) as urlopen,
|
||||||
mock.patch.object(
|
mock.patch.object(
|
||||||
fetch_artists, "read_artist_titles",
|
fetch_artists.ArtistSnapshotUpdater,
|
||||||
|
"_ArtistSnapshotUpdater__read_artist_titles",
|
||||||
side_effect=[[], OSError("boom")])):
|
side_effect=[[], OSError("boom")])):
|
||||||
status: int = self.__run_fetch()[0]
|
status: int = self.__run_fetch()[0]
|
||||||
self.assertNotEqual(status, 0)
|
self.assertNotEqual(status, 0)
|
||||||
|
|||||||
@@ -50,7 +50,9 @@ class TestFetchLyrics(unittest.TestCase):
|
|||||||
self.addCleanup(self.__ds.engine.dispose)
|
self.addCleanup(self.__ds.engine.dispose)
|
||||||
patchers: list[Any] = [
|
patchers: list[Any] = [
|
||||||
mock.patch.object(fetch_lyrics, "ds", self.__ds),
|
mock.patch.object(fetch_lyrics, "ds", self.__ds),
|
||||||
mock.patch.object(fetch_lyrics, "SLEEP_SECONDS", 0.0)]
|
mock.patch.object(
|
||||||
|
fetch_lyrics.LyricsFetcher,
|
||||||
|
"_LyricsFetcher__SLEEP_SECONDS", 0.0)]
|
||||||
for patcher in patchers:
|
for patcher in patchers:
|
||||||
patcher.start()
|
patcher.start()
|
||||||
self.addCleanup(patcher.stop)
|
self.addCleanup(patcher.stop)
|
||||||
@@ -158,8 +160,9 @@ class TestFetchLyrics(unittest.TestCase):
|
|||||||
self.assertEqual(status, 0)
|
self.assertEqual(status, 0)
|
||||||
self.assertEqual(urlopen.call_count, 1)
|
self.assertEqual(urlopen.call_count, 1)
|
||||||
request: Any = urlopen.call_args[0][0]
|
request: Any = urlopen.call_args[0][0]
|
||||||
self.assertEqual(request.get_header("User-agent"),
|
self.assertEqual(
|
||||||
fetch_lyrics.USER_AGENT)
|
request.get_header("User-agent"),
|
||||||
|
fetch_lyrics.LyricsFetcher._LyricsFetcher__USER_AGENT)
|
||||||
self.assertEqual(
|
self.assertEqual(
|
||||||
(self.__lyrics / "1.txt")
|
(self.__lyrics / "1.txt")
|
||||||
.read_text(encoding="utf-8"),
|
.read_text(encoding="utf-8"),
|
||||||
@@ -321,39 +324,39 @@ class TestFetchLyrics(unittest.TestCase):
|
|||||||
def test_normalize_cp1252_mojibake(self) -> None:
|
def test_normalize_cp1252_mojibake(self) -> None:
|
||||||
"""Test that cp1252 mojibake codepoints are restored."""
|
"""Test that cp1252 mojibake codepoints are restored."""
|
||||||
self.assertEqual(
|
self.assertEqual(
|
||||||
fetch_lyrics.normalize_lyrics("wait
"),
|
fetch_lyrics.LyricsFetchRunner.normalize_lyrics("wait
"),
|
||||||
"wait…")
|
"wait…")
|
||||||
self.assertEqual(
|
self.assertEqual(
|
||||||
fetch_lyrics.normalize_lyrics(
|
fetch_lyrics.LyricsFetchRunner.normalize_lyrics(
|
||||||
"quote"),
|
"quote"),
|
||||||
"‘quote’")
|
"‘quote’")
|
||||||
self.assertEqual(
|
self.assertEqual(
|
||||||
fetch_lyrics.normalize_lyrics(
|
fetch_lyrics.LyricsFetchRunner.normalize_lyrics(
|
||||||
"quote"),
|
"quote"),
|
||||||
"“quote”")
|
"“quote”")
|
||||||
self.assertEqual(
|
self.assertEqual(
|
||||||
fetch_lyrics.normalize_lyrics("dashline"),
|
fetch_lyrics.LyricsFetchRunner.normalize_lyrics("dashline"),
|
||||||
"dash—line")
|
"dash—line")
|
||||||
|
|
||||||
def test_normalize_undefined_cp1252_removed(self) -> None:
|
def test_normalize_undefined_cp1252_removed(self) -> None:
|
||||||
"""Test that undefined cp1252 byte values are removed."""
|
"""Test that undefined cp1252 byte values are removed."""
|
||||||
text: str = ("abcdef")
|
text: str = ("abcdef")
|
||||||
self.assertEqual(
|
self.assertEqual(
|
||||||
fetch_lyrics.normalize_lyrics(text), "abcdef")
|
fetch_lyrics.LyricsFetchRunner.normalize_lyrics(text), "abcdef")
|
||||||
|
|
||||||
def test_normalize_homoglyphs(self) -> None:
|
def test_normalize_homoglyphs(self) -> None:
|
||||||
"""Test that watermark homoglyphs are restored."""
|
"""Test that watermark homoglyphs are restored."""
|
||||||
self.assertEqual(
|
self.assertEqual(
|
||||||
fetch_lyrics.normalize_lyrics("likе that"),
|
fetch_lyrics.LyricsFetchRunner.normalize_lyrics("likе that"),
|
||||||
"like that")
|
"like that")
|
||||||
self.assertEqual(
|
self.assertEqual(
|
||||||
fetch_lyrics.normalize_lyrics("lό que soy"),
|
fetch_lyrics.LyricsFetchRunner.normalize_lyrics("lό que soy"),
|
||||||
"ló que soy")
|
"ló que soy")
|
||||||
|
|
||||||
def test_normalize_space_variants(self) -> None:
|
def test_normalize_space_variants(self) -> None:
|
||||||
"""Test that exotic space variants become ASCII space."""
|
"""Test that exotic space variants become ASCII space."""
|
||||||
self.assertEqual(
|
self.assertEqual(
|
||||||
fetch_lyrics.normalize_lyrics(
|
fetch_lyrics.LyricsFetchRunner.normalize_lyrics(
|
||||||
"a b c d"),
|
"a b c d"),
|
||||||
"a b c d")
|
"a b c d")
|
||||||
|
|
||||||
@@ -362,19 +365,19 @@ class TestFetchLyrics(unittest.TestCase):
|
|||||||
text: str = (
|
text: str = (
|
||||||
"abcde")
|
"abcde")
|
||||||
self.assertEqual(
|
self.assertEqual(
|
||||||
fetch_lyrics.normalize_lyrics(text), "abcde")
|
fetch_lyrics.LyricsFetchRunner.normalize_lyrics(text), "abcde")
|
||||||
|
|
||||||
def test_normalize_ascii_unchanged(self) -> None:
|
def test_normalize_ascii_unchanged(self) -> None:
|
||||||
"""Test that plain ASCII text passes through unchanged."""
|
"""Test that plain ASCII text passes through unchanged."""
|
||||||
text: str = "Hello, it's me\n"
|
text: str = "Hello, it's me\n"
|
||||||
self.assertEqual(
|
self.assertEqual(
|
||||||
fetch_lyrics.normalize_lyrics(text), text)
|
fetch_lyrics.LyricsFetchRunner.normalize_lyrics(text), text)
|
||||||
|
|
||||||
def test_normalize_legitimate_non_ascii_unchanged(self) -> None:
|
def test_normalize_legitimate_non_ascii_unchanged(self) -> None:
|
||||||
"""Test that legitimate non-ASCII content is unchanged."""
|
"""Test that legitimate non-ASCII content is unchanged."""
|
||||||
text: str = "¿cómo estás? 안녕하세요\n"
|
text: str = "¿cómo estás? 안녕하세요\n"
|
||||||
self.assertEqual(
|
self.assertEqual(
|
||||||
fetch_lyrics.normalize_lyrics(text), text)
|
fetch_lyrics.LyricsFetchRunner.normalize_lyrics(text), text)
|
||||||
|
|
||||||
def test_fetched_lyrics_saved_normalized(self) -> None:
|
def test_fetched_lyrics_saved_normalized(self) -> None:
|
||||||
"""Test that a fetched lyric is normalized before saving."""
|
"""Test that a fetched lyric is normalized before saving."""
|
||||||
|
|||||||
+113
-169
@@ -90,86 +90,113 @@ class RunLLMTestCase(unittest.TestCase):
|
|||||||
|
|
||||||
|
|
||||||
class TestLoadItems(RunLLMTestCase):
|
class TestLoadItems(RunLLMTestCase):
|
||||||
"""Test cases for the input JSONL validation."""
|
"""Test cases for the input JSONL validation, driven by a dry
|
||||||
|
run since the validation happens before any API call."""
|
||||||
|
|
||||||
def setUp(self) -> None:
|
def setUp(self) -> None:
|
||||||
"""Create a temporary directory for the input files."""
|
"""Create the prompt and input paths for a dry run."""
|
||||||
self.__dir: Path = self._make_temp_dir()
|
directory: Path = self._make_temp_dir()
|
||||||
|
self.__prompt: Path = directory / "task.md"
|
||||||
|
self.__prompt.write_text("The task.\n", encoding="utf-8")
|
||||||
|
self.__input: Path = directory / "items.jsonl"
|
||||||
|
self.__archive_dir: Path = directory / "runs" / "run1"
|
||||||
|
|
||||||
def __write_input(self, content: str) -> Path:
|
def __run_dry(self, content: str) -> tuple[int, str]:
|
||||||
"""Write an input file with the given content.
|
"""Write the input file and dry-run against it.
|
||||||
|
|
||||||
:param content: The file content.
|
:param content: The input file content.
|
||||||
:return: The path of the input file.
|
:return: The exit status and the standard error text.
|
||||||
"""
|
"""
|
||||||
path: Path = self.__dir / "items.jsonl"
|
self.__input.write_text(content, encoding="utf-8")
|
||||||
path.write_text(content, encoding="utf-8")
|
stderr: io.StringIO = io.StringIO()
|
||||||
return path
|
status: int
|
||||||
|
with redirect_stderr(stderr):
|
||||||
def test_valid_items(self) -> None:
|
status = run_llm.main([
|
||||||
"""Test that valid items are loaded in file order."""
|
str(self.__prompt), str(self.__input),
|
||||||
path: Path = self.__write_input(
|
str(self.__archive_dir), "--dry-run"])
|
||||||
'{"id": "a", "content": "one"}\n'
|
return status, stderr.getvalue()
|
||||||
'{"id": "b", "content": "two"}\n')
|
|
||||||
items: list[run_llm.InputItem] = run_llm.load_items(path)
|
|
||||||
self.assertEqual(items, [
|
|
||||||
run_llm.InputItem(id="a", content="one"),
|
|
||||||
run_llm.InputItem(id="b", content="two")])
|
|
||||||
|
|
||||||
def test_malformed_json_names_line(self) -> None:
|
def test_malformed_json_names_line(self) -> None:
|
||||||
"""Test that malformed JSON reports the line number."""
|
"""Test that malformed JSON reports the line number."""
|
||||||
path: Path = self.__write_input(
|
status: int
|
||||||
|
stderr: str
|
||||||
|
status, stderr = self.__run_dry(
|
||||||
'{"id": "a", "content": "one"}\n'
|
'{"id": "a", "content": "one"}\n'
|
||||||
'not json\n')
|
'not json\n')
|
||||||
with self.assertRaises(run_llm.InputFormatError) as context:
|
self.assertEqual(status, 1)
|
||||||
run_llm.load_items(path)
|
self.assertIn("line 2", stderr)
|
||||||
self.assertIn("line 2", str(context.exception))
|
|
||||||
|
|
||||||
def test_missing_key_names_line(self) -> None:
|
def test_missing_key_names_line(self) -> None:
|
||||||
"""Test that a missing key reports the line number."""
|
"""Test that a missing key reports the line number."""
|
||||||
path: Path = self.__write_input('{"id": "a"}\n')
|
status: int
|
||||||
with self.assertRaises(run_llm.InputFormatError) as context:
|
stderr: str
|
||||||
run_llm.load_items(path)
|
status, stderr = self.__run_dry('{"id": "a"}\n')
|
||||||
self.assertIn("line 1", str(context.exception))
|
self.assertEqual(status, 1)
|
||||||
|
self.assertIn("line 1", stderr)
|
||||||
|
|
||||||
def test_extra_key_rejected(self) -> None:
|
def test_extra_key_rejected(self) -> None:
|
||||||
"""Test that an extra key is rejected."""
|
"""Test that an extra key is rejected."""
|
||||||
path: Path = self.__write_input(
|
status: int
|
||||||
|
status, _ = self.__run_dry(
|
||||||
'{"id": "a", "content": "one", "extra": 1}\n')
|
'{"id": "a", "content": "one", "extra": 1}\n')
|
||||||
with self.assertRaises(run_llm.InputFormatError):
|
self.assertEqual(status, 1)
|
||||||
run_llm.load_items(path)
|
|
||||||
|
|
||||||
def test_non_string_content_rejected(self) -> None:
|
def test_non_string_content_rejected(self) -> None:
|
||||||
"""Test that a non-string content is rejected."""
|
"""Test that a non-string content is rejected."""
|
||||||
path: Path = self.__write_input('{"id": "a", "content": 3}\n')
|
status: int
|
||||||
with self.assertRaises(run_llm.InputFormatError):
|
status, _ = self.__run_dry('{"id": "a", "content": 3}\n')
|
||||||
run_llm.load_items(path)
|
self.assertEqual(status, 1)
|
||||||
|
|
||||||
def test_duplicated_id_names_line(self) -> None:
|
def test_duplicated_id_names_line(self) -> None:
|
||||||
"""Test that a duplicated ID reports the line number."""
|
"""Test that a duplicated ID reports the line number."""
|
||||||
path: Path = self.__write_input(
|
status: int
|
||||||
|
stderr: str
|
||||||
|
status, stderr = self.__run_dry(
|
||||||
'{"id": "a", "content": "one"}\n'
|
'{"id": "a", "content": "one"}\n'
|
||||||
'{"id": "a", "content": "two"}\n')
|
'{"id": "a", "content": "two"}\n')
|
||||||
with self.assertRaises(run_llm.InputFormatError) as context:
|
self.assertEqual(status, 1)
|
||||||
run_llm.load_items(path)
|
self.assertIn("line 2", stderr)
|
||||||
self.assertIn("line 2", str(context.exception))
|
self.assertIn("a", stderr)
|
||||||
self.assertIn("a", str(context.exception))
|
|
||||||
|
|
||||||
def test_empty_file_rejected(self) -> None:
|
def test_empty_file_rejected(self) -> None:
|
||||||
"""Test that an empty input file is rejected."""
|
"""Test that an empty input file is rejected."""
|
||||||
path: Path = self.__write_input("")
|
status: int
|
||||||
with self.assertRaises(run_llm.InputFormatError):
|
status, _ = self.__run_dry("")
|
||||||
run_llm.load_items(path)
|
self.assertEqual(status, 1)
|
||||||
|
self.assertFalse(self.__archive_dir.exists())
|
||||||
|
|
||||||
|
|
||||||
class TestRequestBuilding(RunLLMTestCase):
|
class TestRequestBuilding(RunLLMTestCase):
|
||||||
"""Test cases for the request construction."""
|
"""Test cases for the request preview, driven by a dry run."""
|
||||||
|
|
||||||
def test_build_request(self) -> None:
|
def setUp(self) -> None:
|
||||||
"""Test the shape of a batch request."""
|
"""Create the prompt and input files for a dry run."""
|
||||||
request: dict[str, Any] = run_llm.build_request(
|
directory: Path = self._make_temp_dir()
|
||||||
run_llm.InputItem(id="song-1", content="the lyrics"),
|
self.__prompt: Path = directory / "task.md"
|
||||||
"the system prompt", 2048, "claude-sonnet-4-6")
|
self.__prompt.write_text(
|
||||||
|
"the system prompt", encoding="utf-8")
|
||||||
|
self.__input: Path = directory / "items.jsonl"
|
||||||
|
self.__input.write_text(
|
||||||
|
'{"id": "song-1", "content": "the lyrics"}\n',
|
||||||
|
encoding="utf-8")
|
||||||
|
self.__archive_dir: Path = directory / "runs" / "run1"
|
||||||
|
|
||||||
|
def __preview(self, extra_argv: list[str]) -> dict[str, Any]:
|
||||||
|
"""Dry-run and parse the previewed request.
|
||||||
|
|
||||||
|
:param extra_argv: The extra command-line arguments.
|
||||||
|
:return: The parsed request.
|
||||||
|
"""
|
||||||
|
stdout: io.StringIO = io.StringIO()
|
||||||
|
with redirect_stdout(stdout):
|
||||||
|
run_llm.main([
|
||||||
|
str(self.__prompt), str(self.__input),
|
||||||
|
str(self.__archive_dir), "--dry-run"] + extra_argv)
|
||||||
|
return json.loads(stdout.getvalue())
|
||||||
|
|
||||||
|
def test_default_model_request(self) -> None:
|
||||||
|
"""Test the request shape for the default model."""
|
||||||
|
request: dict[str, Any] = self.__preview([])
|
||||||
self.assertEqual(request["custom_id"], "song-1")
|
self.assertEqual(request["custom_id"], "song-1")
|
||||||
params: dict[str, Any] = request["params"]
|
params: dict[str, Any] = request["params"]
|
||||||
self.assertEqual(params["model"], "claude-sonnet-4-6")
|
self.assertEqual(params["model"], "claude-sonnet-4-6")
|
||||||
@@ -177,125 +204,19 @@ class TestRequestBuilding(RunLLMTestCase):
|
|||||||
self.assertEqual(params["thinking"], {"type": "disabled"})
|
self.assertEqual(params["thinking"], {"type": "disabled"})
|
||||||
self.assertEqual(params["max_tokens"], 2048)
|
self.assertEqual(params["max_tokens"], 2048)
|
||||||
self.assertEqual(params["system"], "the system prompt")
|
self.assertEqual(params["system"], "the system prompt")
|
||||||
self.assertEqual(params["messages"],
|
self.assertEqual(
|
||||||
[{"role": "user", "content": "the lyrics"}])
|
params["messages"],
|
||||||
|
[{"role": "user", "content": "the lyrics"}])
|
||||||
|
|
||||||
def test_build_request_fable_5(self) -> None:
|
def test_fable_5_request(self) -> None:
|
||||||
"""Test the request shape for the claude-fable-5 model."""
|
"""Test the request shape for the claude-fable-5 model."""
|
||||||
request: dict[str, Any] = run_llm.build_request(
|
request: dict[str, Any] = self.__preview(
|
||||||
run_llm.InputItem(id="group-1", content="the groups"),
|
["--model", "claude-fable-5", "--max-tokens", "8192"])
|
||||||
"the system prompt", 8192, "claude-fable-5")
|
|
||||||
params: dict[str, Any] = request["params"]
|
params: dict[str, Any] = request["params"]
|
||||||
self.assertEqual(params["model"], "claude-fable-5")
|
self.assertEqual(params["model"], "claude-fable-5")
|
||||||
self.assertNotIn("temperature", params)
|
self.assertNotIn("temperature", params)
|
||||||
self.assertNotIn("thinking", params)
|
self.assertNotIn("thinking", params)
|
||||||
self.assertEqual(params["max_tokens"], 8192)
|
self.assertEqual(params["max_tokens"], 8192)
|
||||||
self.assertEqual(params["system"], "the system prompt")
|
|
||||||
self.assertEqual(params["messages"],
|
|
||||||
[{"role": "user", "content": "the groups"}])
|
|
||||||
|
|
||||||
|
|
||||||
class TestCollectResults(RunLLMTestCase):
|
|
||||||
"""Test cases for the batch result collection."""
|
|
||||||
|
|
||||||
def test_collect_success_and_error(self) -> None:
|
|
||||||
"""Test collecting succeeded and errored results."""
|
|
||||||
client: mock.Mock = mock.Mock()
|
|
||||||
client.messages.batches.results.return_value = iter([
|
|
||||||
self._make_success_entry("a", "output a"),
|
|
||||||
self._make_error_entry("b", "invalid_request_error")])
|
|
||||||
results: run_llm.Results = run_llm.collect_results(
|
|
||||||
client, "batch_x")
|
|
||||||
self.assertEqual(results["a"].text, "output a")
|
|
||||||
self.assertEqual(results["a"].stop_reason, "end_turn")
|
|
||||||
self.assertEqual(results["a"].usage,
|
|
||||||
{"input_tokens": 10, "output_tokens": 5})
|
|
||||||
self.assertEqual(results["b"], run_llm.BatchResult(
|
|
||||||
id="b", error="invalid_request_error"))
|
|
||||||
client.messages.batches.results.assert_called_once_with(
|
|
||||||
"batch_x")
|
|
||||||
|
|
||||||
def test_find_failures(self) -> None:
|
|
||||||
"""Test finding failed and missing items."""
|
|
||||||
results: run_llm.Results = {
|
|
||||||
"a": run_llm.BatchResult(id="a", text="fine"),
|
|
||||||
"b": run_llm.BatchResult(id="b", error="errored")}
|
|
||||||
self.assertEqual(
|
|
||||||
run_llm.find_failures(["a", "b", "c"], results),
|
|
||||||
["b", "c"])
|
|
||||||
|
|
||||||
def test_sum_usage(self) -> None:
|
|
||||||
"""Test summing the token usage across results."""
|
|
||||||
results: run_llm.Results = {
|
|
||||||
"a": run_llm.BatchResult(
|
|
||||||
id="a", text="fine",
|
|
||||||
usage={"input_tokens": 10, "output_tokens": 5}),
|
|
||||||
"b": run_llm.BatchResult(
|
|
||||||
id="b", text="fine",
|
|
||||||
usage={"input_tokens": 3, "output_tokens": 2}),
|
|
||||||
"c": run_llm.BatchResult(id="c", error="errored")}
|
|
||||||
self.assertEqual(
|
|
||||||
run_llm.sum_usage(results),
|
|
||||||
{"input_tokens": 13, "output_tokens": 7})
|
|
||||||
|
|
||||||
|
|
||||||
class TestArchive(RunLLMTestCase):
|
|
||||||
"""Test cases for the archive directory handling."""
|
|
||||||
|
|
||||||
def setUp(self) -> None:
|
|
||||||
"""Create a temporary directory as the runs root."""
|
|
||||||
self.__dir: Path = self._make_temp_dir()
|
|
||||||
|
|
||||||
def test_create_archive_dir(self) -> None:
|
|
||||||
"""Test the archive directory creation."""
|
|
||||||
target: Path = self.__dir / "01-01-tag" / "run1"
|
|
||||||
directory: Path = run_llm.create_archive_dir(target, False)
|
|
||||||
self.assertTrue(directory.is_dir())
|
|
||||||
self.assertEqual(directory, target)
|
|
||||||
|
|
||||||
def test_existing_archive_dir_rejected_without_replace(
|
|
||||||
self) -> None:
|
|
||||||
"""Test that an existing archive is rejected by default."""
|
|
||||||
target: Path = self.__dir / "01-01-tag" / "run1"
|
|
||||||
run_llm.create_archive_dir(target, False)
|
|
||||||
with self.assertRaises(FileExistsError):
|
|
||||||
run_llm.create_archive_dir(target, False)
|
|
||||||
|
|
||||||
def test_existing_archive_dir_replaced(self) -> None:
|
|
||||||
"""Test that --replace replaces an existing archive."""
|
|
||||||
target: Path = self.__dir / "01-01-tag" / "run1"
|
|
||||||
first: Path = run_llm.create_archive_dir(target, False)
|
|
||||||
(first / "stale.txt").write_text("stale", encoding="utf-8")
|
|
||||||
second: Path = run_llm.create_archive_dir(target, True)
|
|
||||||
self.assertEqual(first, second)
|
|
||||||
self.assertFalse((second / "stale.txt").exists())
|
|
||||||
|
|
||||||
def test_replace_leaves_sibling_dir_untouched(self) -> None:
|
|
||||||
"""Test that replacing run2 does not touch run1."""
|
|
||||||
run1: Path = run_llm.create_archive_dir(
|
|
||||||
self.__dir / "01-01-tag" / "run1", False)
|
|
||||||
(run1 / "output.jsonl").write_text(
|
|
||||||
"run1 data", encoding="utf-8")
|
|
||||||
run2: Path = run_llm.create_archive_dir(
|
|
||||||
self.__dir / "01-01-tag" / "run2", False)
|
|
||||||
(run2 / "stale.jsonl").write_text("stale", encoding="utf-8")
|
|
||||||
run_llm.create_archive_dir(
|
|
||||||
self.__dir / "01-01-tag" / "run2", True)
|
|
||||||
self.assertEqual(
|
|
||||||
(run1 / "output.jsonl").read_text(encoding="utf-8"),
|
|
||||||
"run1 data")
|
|
||||||
|
|
||||||
def test_write_jsonl(self) -> None:
|
|
||||||
"""Test writing records as JSON Lines."""
|
|
||||||
path: Path = self.__dir / "out.jsonl"
|
|
||||||
run_llm.write_jsonl(path, [{"id": "a", "text": "中文"},
|
|
||||||
{"id": "b", "text": "two"}])
|
|
||||||
lines: list[str] = path.read_text(
|
|
||||||
encoding="utf-8").splitlines()
|
|
||||||
self.assertEqual(len(lines), 2)
|
|
||||||
self.assertEqual(json.loads(lines[0]),
|
|
||||||
{"id": "a", "text": "中文"})
|
|
||||||
self.assertIn("中文", lines[0])
|
|
||||||
|
|
||||||
|
|
||||||
class TestMainFlow(RunLLMTestCase):
|
class TestMainFlow(RunLLMTestCase):
|
||||||
@@ -322,8 +243,7 @@ class TestMainFlow(RunLLMTestCase):
|
|||||||
ANTHROPIC_API_KEY="test-key")
|
ANTHROPIC_API_KEY="test-key")
|
||||||
config.set_settings(self.__settings)
|
config.set_settings(self.__settings)
|
||||||
|
|
||||||
@staticmethod
|
def __make_client(self, entries: list[Any]) -> mock.Mock:
|
||||||
def __make_client(entries: list[Any]) -> mock.Mock:
|
|
||||||
"""Create a mock Anthropic client serving canned results.
|
"""Create a mock Anthropic client serving canned results.
|
||||||
|
|
||||||
:param entries: The result entries of the single run batch.
|
:param entries: The result entries of the single run batch.
|
||||||
@@ -386,9 +306,11 @@ class TestMainFlow(RunLLMTestCase):
|
|||||||
"Done. 2 jobs finished. 02:05 elapsed."))
|
"Done. 2 jobs finished. 02:05 elapsed."))
|
||||||
|
|
||||||
def test_run_produces_output_file(self) -> None:
|
def test_run_produces_output_file(self) -> None:
|
||||||
"""Test that a run submits one batch and writes output."""
|
"""Test that a run submits one batch and writes output,
|
||||||
|
the item order preserved and the token usage summed, the
|
||||||
|
output text written unescaped."""
|
||||||
client: mock.Mock = self.__make_client(
|
client: mock.Mock = self.__make_client(
|
||||||
[self._make_success_entry("a", "answer a"),
|
[self._make_success_entry("a", "answer 中文 a"),
|
||||||
self._make_success_entry("b", "answer b")])
|
self._make_success_entry("b", "answer b")])
|
||||||
status: int
|
status: int
|
||||||
stderr: str
|
stderr: str
|
||||||
@@ -400,10 +322,13 @@ class TestMainFlow(RunLLMTestCase):
|
|||||||
client.messages.batches.create.call_count, 1)
|
client.messages.batches.create.call_count, 1)
|
||||||
run_dir: Path = self.__archive_dir
|
run_dir: Path = self.__archive_dir
|
||||||
self.assertTrue((run_dir / "output.jsonl").exists())
|
self.assertTrue((run_dir / "output.jsonl").exists())
|
||||||
|
output_text: str = (run_dir / "output.jsonl").read_text(
|
||||||
|
encoding="utf-8")
|
||||||
|
self.assertIn("中文", output_text)
|
||||||
output: list[dict[str, Any]] = [
|
output: list[dict[str, Any]] = [
|
||||||
json.loads(x) for x in (run_dir / "output.jsonl")
|
json.loads(x) for x in output_text.splitlines()]
|
||||||
.read_text(encoding="utf-8").splitlines()]
|
self.assertEqual(output[0]["text"], "answer 中文 a")
|
||||||
self.assertEqual(output[0]["text"], "answer a")
|
self.assertEqual(output[1]["text"], "answer b")
|
||||||
meta: dict[str, Any] = json.loads(
|
meta: dict[str, Any] = json.loads(
|
||||||
(run_dir / "meta.json").read_text(encoding="utf-8"))
|
(run_dir / "meta.json").read_text(encoding="utf-8"))
|
||||||
self.assertNotIn("run", meta)
|
self.assertNotIn("run", meta)
|
||||||
@@ -465,11 +390,14 @@ class TestMainFlow(RunLLMTestCase):
|
|||||||
"stale run2 data")
|
"stale run2 data")
|
||||||
|
|
||||||
def test_run_failure_exits_non_zero(self) -> None:
|
def test_run_failure_exits_non_zero(self) -> None:
|
||||||
"""Test that a failed item aborts with a non-zero status."""
|
"""Test that a failed item aborts with a non-zero status,
|
||||||
|
the summed usage counting only the succeeded item."""
|
||||||
client: mock.Mock = self.__make_client(
|
client: mock.Mock = self.__make_client(
|
||||||
[self._make_success_entry("a", "answer a"),
|
[self._make_success_entry("a", "answer a"),
|
||||||
self._make_error_entry("b", "invalid_request_error")])
|
self._make_error_entry("b", "invalid_request_error")])
|
||||||
status: int = self.__run_main(self.__argv, client)[0]
|
status: int
|
||||||
|
stderr: str
|
||||||
|
status, _, stderr = self.__run_main(self.__argv, client)
|
||||||
self.assertEqual(status, 1)
|
self.assertEqual(status, 1)
|
||||||
run_dir: Path = self.__archive_dir
|
run_dir: Path = self.__archive_dir
|
||||||
self.assertTrue((run_dir / "output.jsonl").exists())
|
self.assertTrue((run_dir / "output.jsonl").exists())
|
||||||
@@ -479,6 +407,22 @@ class TestMainFlow(RunLLMTestCase):
|
|||||||
self.assertEqual(json.loads(output_lines[1]),
|
self.assertEqual(json.loads(output_lines[1]),
|
||||||
{"id": "b",
|
{"id": "b",
|
||||||
"error": "invalid_request_error"})
|
"error": "invalid_request_error"})
|
||||||
|
meta: dict[str, Any] = json.loads(
|
||||||
|
(run_dir / "meta.json").read_text(encoding="utf-8"))
|
||||||
|
self.assertEqual(meta["usage"],
|
||||||
|
{"input_tokens": 10, "output_tokens": 5})
|
||||||
|
self.assertNotIn("Done.", stderr)
|
||||||
|
|
||||||
|
def test_missing_result_item_is_a_failure(self) -> None:
|
||||||
|
"""Test that an item missing from the batch results is
|
||||||
|
reported as a failed item."""
|
||||||
|
client: mock.Mock = self.__make_client(
|
||||||
|
[self._make_success_entry("a", "answer a")])
|
||||||
|
status: int
|
||||||
|
stderr: str
|
||||||
|
status, _, stderr = self.__run_main(self.__argv, client)
|
||||||
|
self.assertEqual(status, 1)
|
||||||
|
self.assertIn("b", stderr)
|
||||||
|
|
||||||
def test_invalid_input_exits_non_zero(self) -> None:
|
def test_invalid_input_exits_non_zero(self) -> None:
|
||||||
"""Test that an invalid input file aborts before archiving."""
|
"""Test that an invalid input file aborts before archiving."""
|
||||||
|
|||||||
@@ -11,12 +11,14 @@ import tempfile
|
|||||||
import unittest
|
import unittest
|
||||||
from contextlib import redirect_stderr
|
from contextlib import redirect_stderr
|
||||||
from pathlib import Path
|
from pathlib import Path
|
||||||
|
from typing import Any
|
||||||
|
from unittest import mock
|
||||||
|
|
||||||
import sqlalchemy as sa
|
|
||||||
from sqlalchemy.orm import Session
|
from sqlalchemy.orm import Session
|
||||||
|
|
||||||
|
from pop_fem_audit_tools import config
|
||||||
from pop_fem_audit_tools.commands import tally_annotations
|
from pop_fem_audit_tools.commands import tally_annotations
|
||||||
from pop_fem_audit_tools.database import Base
|
from pop_fem_audit_tools.database import Base, DataSource
|
||||||
from pop_fem_audit_tools.models import Song
|
from pop_fem_audit_tools.models import Song
|
||||||
|
|
||||||
|
|
||||||
@@ -24,8 +26,8 @@ class TestTallyAnnotations(unittest.TestCase):
|
|||||||
"""Test cases for the step-5 pattern annotation tally."""
|
"""Test cases for the step-5 pattern annotation tally."""
|
||||||
|
|
||||||
def setUp(self) -> None:
|
def setUp(self) -> None:
|
||||||
"""Create the archive directories, the database, and the
|
"""Create the archive directories, the working store, and
|
||||||
output paths."""
|
the output paths."""
|
||||||
tmp: tempfile.TemporaryDirectory[str] \
|
tmp: tempfile.TemporaryDirectory[str] \
|
||||||
= tempfile.TemporaryDirectory()
|
= tempfile.TemporaryDirectory()
|
||||||
self.addCleanup(tmp.cleanup)
|
self.addCleanup(tmp.cleanup)
|
||||||
@@ -37,7 +39,6 @@ class TestTallyAnnotations(unittest.TestCase):
|
|||||||
self.__male_synthesis.mkdir()
|
self.__male_synthesis.mkdir()
|
||||||
self.__female_synthesis.mkdir()
|
self.__female_synthesis.mkdir()
|
||||||
self.__mixed_synthesis.mkdir()
|
self.__mixed_synthesis.mkdir()
|
||||||
self.__db_path: Path = self.__dir / "working.sqlite3"
|
|
||||||
self.__patterns_csv: Path \
|
self.__patterns_csv: Path \
|
||||||
= self.__dir / "results" / "patterns.csv"
|
= self.__dir / "results" / "patterns.csv"
|
||||||
self.__annotations_csv: Path \
|
self.__annotations_csv: Path \
|
||||||
@@ -48,6 +49,15 @@ class TestTallyAnnotations(unittest.TestCase):
|
|||||||
run_dir: Path = self.__dir / f"run{number}"
|
run_dir: Path = self.__dir / f"run{number}"
|
||||||
run_dir.mkdir()
|
run_dir.mkdir()
|
||||||
self.__runs.append(run_dir)
|
self.__runs.append(run_dir)
|
||||||
|
config.set_settings(config.Settings(
|
||||||
|
SQLALCHEMY_DATABASE_URL="sqlite://",
|
||||||
|
ANTHROPIC_API_KEY="test-key"))
|
||||||
|
self.__ds: DataSource = DataSource()
|
||||||
|
self.addCleanup(self.__ds.engine.dispose)
|
||||||
|
patcher: Any = mock.patch.object(
|
||||||
|
tally_annotations, "ds", self.__ds)
|
||||||
|
patcher.start()
|
||||||
|
self.addCleanup(patcher.stop)
|
||||||
|
|
||||||
def __write_default_synthesis_archives(self) -> None:
|
def __write_default_synthesis_archives(self) -> None:
|
||||||
"""Write the male, female, and mixed synthesis archives.
|
"""Write the male, female, and mixed synthesis archives.
|
||||||
@@ -108,22 +118,20 @@ class TestTallyAnnotations(unittest.TestCase):
|
|||||||
stored performer gender of every fixture song.
|
stored performer gender of every fixture song.
|
||||||
:return: None.
|
:return: None.
|
||||||
"""
|
"""
|
||||||
engine: sa.Engine = sa.create_engine(
|
Base.metadata.create_all(self.__ds.engine)
|
||||||
f"sqlite:///{self.__db_path}")
|
session: Session = self.__ds.get_db()
|
||||||
Base.metadata.create_all(engine)
|
try:
|
||||||
session: Session
|
|
||||||
with Session(engine) as session:
|
|
||||||
song_id: int
|
song_id: int
|
||||||
title: str
|
title: str
|
||||||
artist_credit: str
|
artist_credit: str
|
||||||
gender: str | None
|
gender: str | None
|
||||||
for song_id, title, artist_credit, gender in songs:
|
for song_id, title, artist_credit, gender in songs:
|
||||||
session.add(Song(
|
session.add(Song(
|
||||||
id=song_id, title=title,
|
id=song_id, title=title, artist_credit=artist_credit,
|
||||||
artist_credit=artist_credit,
|
|
||||||
performer_gender=gender))
|
performer_gender=gender))
|
||||||
session.commit()
|
session.commit()
|
||||||
engine.dispose()
|
finally:
|
||||||
|
session.close()
|
||||||
|
|
||||||
def __write_run(
|
def __write_run(
|
||||||
self, run_dir: Path,
|
self, run_dir: Path,
|
||||||
@@ -165,11 +173,10 @@ class TestTallyAnnotations(unittest.TestCase):
|
|||||||
status: int
|
status: int
|
||||||
with redirect_stderr(stderr):
|
with redirect_stderr(stderr):
|
||||||
status = tally_annotations.main([
|
status = tally_annotations.main([
|
||||||
str(self.__male_synthesis),
|
"--female", str(self.__female_synthesis),
|
||||||
str(self.__female_synthesis),
|
"--male", str(self.__male_synthesis),
|
||||||
str(self.__mixed_synthesis), str(self.__db_path),
|
"--mixed", str(self.__mixed_synthesis),
|
||||||
str(self.__patterns_csv),
|
str(self.__patterns_csv), str(self.__annotations_csv)]
|
||||||
str(self.__annotations_csv)]
|
|
||||||
+ [str(x) for x in self.__runs])
|
+ [str(x) for x in self.__runs])
|
||||||
return status, stderr.getvalue()
|
return status, stderr.getvalue()
|
||||||
|
|
||||||
@@ -315,11 +322,10 @@ class TestTallyAnnotations(unittest.TestCase):
|
|||||||
status: int
|
status: int
|
||||||
with redirect_stderr(stderr):
|
with redirect_stderr(stderr):
|
||||||
status = tally_annotations.main([
|
status = tally_annotations.main([
|
||||||
str(self.__male_synthesis),
|
"--female", str(self.__female_synthesis),
|
||||||
str(self.__female_synthesis),
|
"--male", str(self.__male_synthesis),
|
||||||
str(self.__mixed_synthesis), str(self.__db_path),
|
"--mixed", str(self.__mixed_synthesis),
|
||||||
str(self.__patterns_csv),
|
str(self.__patterns_csv), str(self.__annotations_csv)]
|
||||||
str(self.__annotations_csv)]
|
|
||||||
+ [str(x) for x in self.__runs]
|
+ [str(x) for x in self.__runs]
|
||||||
+ [str(rescue_dir)])
|
+ [str(rescue_dir)])
|
||||||
self.assertEqual(status, 0)
|
self.assertEqual(status, 0)
|
||||||
@@ -329,6 +335,40 @@ class TestTallyAnnotations(unittest.TestCase):
|
|||||||
["Song", "Artist Credit", "Pattern", "Votes"],
|
["Song", "Artist Credit", "Pattern", "Votes"],
|
||||||
["Song A", "Artist A", "M1", "3"]])
|
["Song A", "Artist A", "M1", "3"]])
|
||||||
|
|
||||||
|
def test_dead_record_with_empty_text_rescued_from_extra_run_dir(
|
||||||
|
self) -> None:
|
||||||
|
"""Test that a dead record whose "text" is an empty
|
||||||
|
string (malformed JSON) in one run directory is skipped
|
||||||
|
with a warning, and its ballot is rescued from an extra
|
||||||
|
run directory, tallying successfully."""
|
||||||
|
self.__write_default_synthesis_archives()
|
||||||
|
self.__seed_songs([(1, "Song A", "Artist A", "male")])
|
||||||
|
self.__write_same_ballots_to_all_runs({1: ["M1"]})
|
||||||
|
(self.__runs[0] / "output.jsonl").write_text(
|
||||||
|
json.dumps({"id": "song-1", "text": ""}) + "\n",
|
||||||
|
encoding="utf-8")
|
||||||
|
rescue_dir: Path = self.__dir / "rescue"
|
||||||
|
rescue_dir.mkdir()
|
||||||
|
self.__write_run(rescue_dir, {1: ["M1"]})
|
||||||
|
stderr: io.StringIO = io.StringIO()
|
||||||
|
status: int
|
||||||
|
with redirect_stderr(stderr):
|
||||||
|
status = tally_annotations.main([
|
||||||
|
"--female", str(self.__female_synthesis),
|
||||||
|
"--male", str(self.__male_synthesis),
|
||||||
|
"--mixed", str(self.__mixed_synthesis),
|
||||||
|
str(self.__patterns_csv), str(self.__annotations_csv)]
|
||||||
|
+ [str(x) for x in self.__runs]
|
||||||
|
+ [str(rescue_dir)])
|
||||||
|
self.assertEqual(status, 0)
|
||||||
|
self.assertIn("warning:", stderr.getvalue())
|
||||||
|
self.assertIn("song-1", stderr.getvalue())
|
||||||
|
self.assertIn("\"text\" is malformed JSON", stderr.getvalue())
|
||||||
|
self.assertIn("skipped", stderr.getvalue())
|
||||||
|
self.assertEqual(self.__read_rows(self.__annotations_csv), [
|
||||||
|
["Song", "Artist Credit", "Pattern", "Votes"],
|
||||||
|
["Song A", "Artist A", "M1", "3"]])
|
||||||
|
|
||||||
def test_unknown_pattern_id_dropped_with_warning(self) -> None:
|
def test_unknown_pattern_id_dropped_with_warning(self) -> None:
|
||||||
"""Test that a selected ID outside the extracted patterns
|
"""Test that a selected ID outside the extracted patterns
|
||||||
is dropped, the occurrence reported on standard error."""
|
is dropped, the occurrence reported on standard error."""
|
||||||
@@ -511,7 +551,7 @@ class TestTallyAnnotations(unittest.TestCase):
|
|||||||
|
|
||||||
def test_summary_line_reports_counts(self) -> None:
|
def test_summary_line_reports_counts(self) -> None:
|
||||||
"""Test that the closing summary reports the settled
|
"""Test that the closing summary reports the settled
|
||||||
pair, song, and dropped vote counts."""
|
annotation count."""
|
||||||
self.__write_default_synthesis_archives()
|
self.__write_default_synthesis_archives()
|
||||||
self.__seed_songs([(1, "Song A", "Artist A", "male")])
|
self.__seed_songs([(1, "Song A", "Artist A", "male")])
|
||||||
self.__write_same_ballots_to_all_runs(
|
self.__write_same_ballots_to_all_runs(
|
||||||
@@ -520,6 +560,4 @@ class TestTallyAnnotations(unittest.TestCase):
|
|||||||
stderr: str
|
stderr: str
|
||||||
status, stderr = self.__run_tally()
|
status, stderr = self.__run_tally()
|
||||||
self.assertEqual(status, 0)
|
self.assertEqual(status, 0)
|
||||||
self.assertIn("Tallied 1 settled pairs across 1 songs",
|
self.assertIn("Tallied 1 annotations.", stderr)
|
||||||
stderr)
|
|
||||||
self.assertIn("3 votes dropped", stderr)
|
|
||||||
|
|||||||
Some files were not shown because too many files have changed in this diff Show More
Reference in New Issue
Block a user