SEA-LION-Pile is the pretraining data set for SEA-LION, a collection of Large Language Models (LLMs) which has been pretrained and instruct-tuned for the Southeast Asia (SEA) region. This repository contains the cleaned mC4 portion of the SEA-LION-Pile.
For the remainder of the SEA-LION-Pile dataset, they may be downloaded from the links provided below.
SEA-LION was trained on 980B tokens of the following data:
| Data Source | Unique Tokens | Multiplier | Total Tokens | Percentage |
|---|---|---|---|---|
| RefinedWeb - English | 571.3B | 1 | 571.3B | 58.20% |
| mC4 - Chinese | 91.2B | 1 | 91.2B | 9.29% |
| mC4 - Indonesian | 3.68B | 4 | 14.7B | 1.50% |
| mC4 - Malay | 0.72B | 4 | 2.9B | 0.29% |
| mC4 - Filipino | 1.32B | 4 | 5.3B | 0.54% |
| mC4 - Burmese | 1.2B | 4 | 4.9B | 0.49% |
| mC4 - Vietnamese | 63.4B | 1 | 63.4B | 6.46% |
| mC4 - Thai | 5.8B | 2 | 11.6B | 1.18% |
| WangChanBERTa - Thai | 5B | 2 | 10B | 1.02% |
| mC4 - Lao | 0.27B | 4 | 1.1B | 0.12% |
| mC4 - Khmer | 0.97B | 4 | 3.9B | 0.40% |
| mC4 - Tamil | 2.55B | 4 | 10.2B | 1.04% |
| the Stack - Python | 20.9B | 2 | 41.8B | 4.26% |
| the Stack - Javascript | 55.6B | 1 | 55.6B | 5.66% |
| the Stack - Shell | 1.25B | 2 | 2.5B | 0.26% |
| the Stack - SQL | 6.4B | 2 | 12.8B | 1.31% |
| the Stack - Markdown | 26.6B | 1 | 26.6B | 2.71% |
| RedPajama - StackExchange | 21.2B | 1 | 21.2B | 2.16% |
| RedPajama - ArXiv | 30.6B | 1 | 30.6B | 3.12% |
This section contains the links to the additional datasets that form the SEA-LION-Pile.
This public extract of mC4 is made available under ODC-By 1.0 license; users should also abide to the CommonCrawl ToU.
For all other licenses, please refer to their individual pages above.
We endeavor to ensure data used is permissible and have chosen datasets from creators who have processes to exclude copyrighted or disputed data. For other new data, we have obtained permission to use and distribute.
@misc{lowphansirikul2021wangchanberta,
title={WangchanBERTa: Pretraining transformer-based Thai Language Models},
author={Lalita Lowphansirikul and Charin Polpanumas and Nawat Jantrakulchai and Sarana Nutanong},
year={2021},
eprint={2101.09635},
archivePrefix={arXiv},
primaryClass={cs.CL}
}
@article{refinedweb,
title={The {R}efined{W}eb dataset for {F}alcon {LLM}: outperforming curated corpora with web data, and web data only},
author={Guilherme Penedo and Quentin Malartic and Daniel Hesslow and Ruxandra Cojocaru and Alessandro Cappelli and Hamza Alobeidli and Baptiste Pannier and Ebtesam Almazrouei and Julien Launay},
journal={arXiv preprint arXiv:2306.01116},
eprint={2306.01116},
eprinttype = {arXiv},
url={https://arxiv.org/abs/2306.01116},
year={2023}
}
@article{Kocetkov2022TheStack,
title={The Stack: 3 TB of permissively licensed source code},
author={Kocetkov, Denis and Li, Raymond and Ben Allal, Loubna and Li, Jia and Mou,Chenghao and Muñoz Ferrandis, Carlos and Jernite, Yacine and Mitchell, Margaret and Hughes, Sean and Wolf, Thomas and Bahdanau, Dzmitry and von Werra, Leandro and de Vries, Harm},
journal={Preprint},
year={2022}
}
@software{together2023redpajama,
author = {Together Computer},
title = {RedPajama: An Open Source Recipe to Reproduce LLaMA training dataset},
month = April,
year = 2023,
url = {https://github.com/togethercomputer/RedPajama-Data}
}
SEA-LION-Pile is the pretraining data set for SEA-LION, a collection of Large Language Models (LLMs) which has been pretrained and instruct-tuned for the Southeast Asia (SEA) region. This repository contains the cleaned mC4 portion of the SEA-LION-Pile.
For the remainder of the SEA-LION-Pile dataset, they may be downloaded from the links provided below.
SEA-LION was trained on 980B tokens of the following data:
| Data Source | Unique Tokens | Multiplier | Total Tokens | Percentage |
|---|---|---|---|---|
| RefinedWeb - English | 571.3B | 1 | 571.3B | 58.20% |
| mC4 - Chinese | 91.2B | 1 | 91.2B | 9.29% |
| mC4 - Indonesian | 3.68B | 4 | 14.7B | 1.50% |
| mC4 - Malay | 0.72B | 4 | 2.9B | 0.29% |
| mC4 - Filipino | 1.32B | 4 | 5.3B | 0.54% |
| mC4 - Burmese | 1.2B | 4 | 4.9B | 0.49% |
| mC4 - Vietnamese | 63.4B | 1 | 63.4B | 6.46% |
| mC4 - Thai | 5.8B | 2 | 11.6B | 1.18% |
| WangChanBERTa - Thai | 5B | 2 | 10B | 1.02% |
| mC4 - Lao | 0.27B | 4 | 1.1B | 0.12% |
| mC4 - Khmer | 0.97B | 4 | 3.9B | 0.40% |
| mC4 - Tamil | 2.55B | 4 | 10.2B | 1.04% |
| the Stack - Python | 20.9B | 2 | 41.8B | 4.26% |
| the Stack - Javascript | 55.6B | 1 | 55.6B | 5.66% |
| the Stack - Shell | 1.25B | 2 | 2.5B | 0.26% |
| the Stack - SQL | 6.4B | 2 | 12.8B | 1.31% |
| the Stack - Markdown | 26.6B | 1 | 26.6B | 2.71% |
| RedPajama - StackExchange | 21.2B | 1 | 21.2B | 2.16% |
| RedPajama - ArXiv | 30.6B | 1 | 30.6B | 3.12% |
This section contains the links to the additional datasets that form the SEA-LION-Pile.
This public extract of mC4 is made available under ODC-By 1.0 license; users should also abide to the CommonCrawl ToU.
For all other licenses, please refer to their individual pages above.
We endeavor to ensure data used is permissible and have chosen datasets from creators who have processes to exclude copyrighted or disputed data. For other new data, we have obtained permission to use and distribute.
@misc{lowphansirikul2021wangchanberta,
title={WangchanBERTa: Pretraining transformer-based Thai Language Models},
author={Lalita Lowphansirikul and Charin Polpanumas and Nawat Jantrakulchai and Sarana Nutanong},
year={2021},
eprint={2101.09635},
archivePrefix={arXiv},
primaryClass={cs.CL}
}
@article{refinedweb,
title={The {R}efined{W}eb dataset for {F}alcon {LLM}: outperforming curated corpora with web data, and web data only},
author={Guilherme Penedo and Quentin Malartic and Daniel Hesslow and Ruxandra Cojocaru and Alessandro Cappelli and Hamza Alobeidli and Baptiste Pannier and Ebtesam Almazrouei and Julien Launay},
journal={arXiv preprint arXiv:2306.01116},
eprint={2306.01116},
eprinttype = {arXiv},
url={https://arxiv.org/abs/2306.01116},
year={2023}
}
@article{Kocetkov2022TheStack,
title={The Stack: 3 TB of permissively licensed source code},
author={Kocetkov, Denis and Li, Raymond and Ben Allal, Loubna and Li, Jia and Mou,Chenghao and Muñoz Ferrandis, Carlos and Jernite, Yacine and Mitchell, Margaret and Hughes, Sean and Wolf, Thomas and Bahdanau, Dzmitry and von Werra, Leandro and de Vries, Harm},
journal={Preprint},
year={2022}
}
@software{together2023redpajama,
author = {Together Computer},
title = {RedPajama: An Open Source Recipe to Reproduce LLaMA training dataset},
month = April,
year = 2023,
url = {https://github.com/togethercomputer/RedPajama-Data}
}