@inproceedings{wei-etal-2025-interrogating,
author = {Tian-Zheng Wei, Johnny and Wang, Maggie and Godbole, Ameya and Choi, Jonathan and Jia, Robin},
title = {Interrogating LLM design under copyright law},
year = {2025},
isbn = {9798400714825},
publisher = {Association for Computing Machinery},
address = {New York, NY, USA},
url = {https://doi.org/10.1145/3715275.3732193},
doi = {10.1145/3715275.3732193},
abstract = {The current discourse on large language models (LLMs) and copyright largely takes a “behavioral” perspective, focusing on model outputs and evaluating whether they are substantially similar to training data. However, substantial similarity is difficult to define algorithmically and a narrow focus on model outputs is insufficient to address all copyright risks. In this interdisciplinary work, we take a complementary “structural” perspective and shift our focus to how LLMs are trained. We operationalize a notion of “fair learning” by measuring whether any training decision substantially affected the model’s memorization. As a case study, we deconstruct Pythia, an open-source LLM, and demonstrate the use of causal and correlational analyses to make factual determinations about Pythia’s training decisions. By proposing a legal standard for fair learning and connecting memorization analyses to this standard, we identify how judges may advance the goals of copyright law through adjudication. Finally, we discuss how a fair learning standard might evolve to enhance its clarity by becoming more rule-like and incorporating external technical guidelines.},
booktitle = {Proceedings of the 2025 ACM Conference on Fairness, Accountability, and Transparency},
pages = {3030–3045},
numpages = {16},
keywords = {Copyright, LLMs, regression},
location = {
},
series = {FAccT '25}
}
