@article{laion5b2022, title = {LAION-5B (open image-text training dataset)}, author = {Schuhmann, C. and Beaumont, R. and Vencu, R. and Gordon, C. and Wightman, R. and Cherti, M. and Coombes, T. and Katta, A. and Mullis, C. and Wortsman, M. and et al.}, year = {2022}, journal = {NeurIPS 2022 Datasets and Benchmarks}, url = {https://arxiv.org/abs/2210.08402}, abstract = {Schuhmann et al. (2022) released LAION-5B, an open dataset of approximately 5.85 billion image-text pairs assembled by filtering Common Crawl web data using CLIP similarity scores. Stable Diffusion and the majority of open generative image models were trained on LAION-5B subsets, making it the concrete training-data substrate beneath the open generative-AI ecosystem. The dataset is named in the Andersen v Stability AI copyright litigation as the dataset at issue, creating a direct link between the technical and legal entries in the repository. Teaching data provenance, consent, scraping ethics and copyright exposure in AI image generation requires understanding LAION-5B specifically.}, keywords = {training-data, ip-and-copyright, generative-ai}, note = {AI \& Animation Education Knowledge Base} }