data_tooling
Tools for managing datasets for governance and training.
File Explorer
Download Latest Version (.zip)- add-issue-to-project.yml
- label-with-contact-neede.yml
- label-with-help-wanted.yml
- pii-manager.yml
- self-assign.yaml
- self_deduplicate_ar.yaml
- self_deduplicate_bn.yaml
- self_deduplicate_ca.yaml
- self_deduplicate_en.yaml
- self_deduplicate_es.yaml
- self_deduplicate_eu.yaml
- self_deduplicate_fr.yaml
- self_deduplicate_gl.yaml
- self_deduplicate_hi.yaml
- self_deduplicate_id.yaml
- self_deduplicate_pt.yaml
- self_deduplicate_ur.yaml
- self_deduplicate_vi.yaml
- self_deduplicate_zh.yaml
- __init__.py
- util.py
- README.md
- self_deduplicate.py
- visualize.ipynb
- get_data_for_visualization.py
- README.md
- visualization.py
- anonymization.py
- download_sentencepiece_kenlm_models.py
- explanation_filtering_pipeline.pdf
- filtering.py
- flagged_words.py
- languages_id.py
- main_filtering.py
- muliwai
- normalization.py
- parameters_filtering.py
- person_and_id_anonymization.py
- README.md
- stopwords.py
- test_anonymization.py
- config.json
- tokenizer.json
- config.json
- tokenizer.json
- paws.yaml
- run_glue.py
- run_ner.ipynb
- run_ner.py
- token.yaml
- xnli.yaml
- bertin-tilt.png
- bertin.png
- ccnet.png
- datasets-perp-20-120.png
- datasets-perp.png
- datasets-random-comparison.png
- datasets-wsize.png
- perp-p95.png
- perp-resample-gaussian.png
- perp-resample-stepwise.png
- perplexity_colored_embeddings.html
- random_512.jpg
- dummy_data.zip
- mc4.py
- README.md
- dataset_perplexity.py
- download_mc4es_sampled.py
- generate_datasets.py
- config.json
- config.py
- convert.py
- events.out.tfevents.1625704081.t1v-n-a4d97d44-w-0.212075.3.v2
- events.out.tfevents.1625704245.t1v-n-a4d97d44-w-0.216676.3.v2
- events.out.tfevents.1625705283.t1v-n-a4d97d44-w-0.234462.3.v2
- get_embeddings_and_perplexity.py
- merges.txt
- perplexity.py
- README.md
- run.sh
- run_mlm_flax.py
- run_mlm_flax_stream.py
- run_stream.sh
- special_tokens_map.json
- tokenizer.json
- tokenizer_config.json
- tokens.py
- tokens.py.orig
- tsne_plot.py
- vocab.json
- annotate_langid_crawl.py
- check_wrong_files.py
- compute_stats_langid.py
- detect_html_lang_attrib.py
- 02_detect_html_lang_attrib.slurm
- job_annotate_langid_crawl.sh
- NigerCongoDS.ipynb
- pseudocrawl_nigercongo.ipynb
- extract_text_and_html_metadata.py
- requirements.txt
- cc_lookup_next.py
- cc_lookup_seed.py
- check_erros_in_dataset.py
- deeper.py
- divide_in_shards.py
- download_warc.py
- exact_deduplicates.py
- finalise.py
- load_all_seed_ids.py
- merge_seed_shards.py
- preprocess_dataset.py
- process_for_concatenation.py
- pseudo_crawl_seed_to_lm_dset.py
- pseudo_crawl_seed_to_lm_dset_v2.py
- redownload_warc.py
- requirements.txt
- shard_and_compress.py
- shard_by_seed_id.py
- check_errors_in_dataset.slurm
- divide_in_subshards.slurm
- divide_in_subshards_1000.slurm
- download_warc.slurm
- download_warc_too_big.slurm
- download_warc_trial_4.slurm
- download_warc_trial_5.slurm
- extract_text_and_html_metadata.slurm
- merge_seed_shards.slurm
- preprocess_warc.slurm
- redownload_warc.slurm
- shard_and_compress.slurm
- shard_by_seed_id.slurm
- candidate_websites_for_crawling.csv
- cc-metrics.csv
- cc-metrics.ipynb
- cleanup-seeds.ipynb
- filtered_catalogue.json
- preprocess_dataset.ipynb
- README.md
- seeds.csv
- test_preprcessing_via_pyarrow_pandas.ipynb
- .gitignore
- DEPTH.md
- README.md
- 00_clean_dataset.slurm
- 01_exact_deduplicates.slurm
- 01_download_warc.slurm
- 02_redownload_warc.slurm
- 02b_redownload_warc.slurm
- 03_check_errors_in_dataset.slurm
- 04_divide_in_subshards.slurm
- 05_preprocess_warc.slurm
- 06_extract_text_and_html_metadata.slurm
- 07_shard_by_seed_id.slurm
- 08_merge_seed_shards.slurm
- 09_shard_and_compress.slurm
- 10_push_to_hub.slurm
- cleanup-seeds.ipynb
- seeds.csv
- seeds_batch_2.csv
- seeds_batch_2.json
- .gitignore
- README.md
- get_stats.py
- datasets_ES_builder.py
- datasets_ES_index.py
- datasets_ES_search.py
- datasets_remote_ES_IBMcloud.py
- docker-compose.yml
- README.md
- requirements.txt
- cutoff.csv
- test_stats.json
- __init__.py
- dl_cc_100.py
- expand_corpus.py
- make_dmoz_corpus.py
- __init__.py
- __main__.py
- dedup.py
- execution.py
- flat_hash_set.py
- get_hf_dataset.py
- get_wiki_cirrus.py
- jsonql.py
- mine.py
- minify.py
- perplexity.py
- process_wet_file.py
- regroup.py
- split_by_lang.py
- text_normalizer.py
- tokenizer.py
- lid_exp.json
- mine_segment.json
- test_reproduce.json
- test_segment.json
- sample.warc.txt
- __init__.py
- conftest.py
- test_dedup.py
- test_flat_hash_set.py
- test_jsonql.py
- test_minify.py
- test_normalizer.py
- test_parse_wet_file.py
- test_regroup.py
- test_transformer.py
- .gitignore
- LICENSE
- Makefile
- pyproject.toml
- README.md
- setup.py
- train_all.sh
- __init__.py
- data.py
- engine.py
- perplexity.py
- visualization.py
- __init__.py
- test_data.py
- app.py
- cli.py
- poetry.lock
- pyproject.toml
- README.md
- requirements.txt
- contributing.md
- external.md
- tasks.md
- usage.md
- __init__.py
- file.py
- manager.py
- __init__.py
- manage.py
- task_info.py
- __init__.py
- base.py
- context.py
- exception.py
- json.py
- normalizer.py
- taskdict.py
- types.py
- __init__.py
- bitcoin_address.py
- credit_card.py
- email.py
- ip_address.py
- __init__.py
- international_phone_number.py
- __init__.py
- abn.py
- tfn.py
- __init__.py
- social_insurance_number.py
- __init__.py
- aadhaar.py
- __init__.py
- social_security_number.py
- __init__.py
- __init__.py
- international_phone_number.py
- __init__.py
- bank_account.py
- govid.py
- __init__.py
- curp.py
- __init__.py
- __init__.py
- social_insurance_number.py
- __init__.py
- __init__.py
- cpf.py
- __init__.py
- govid.py
- __init__.py
- __init__.py
- gov_id.py
- misc.py
- __init__.py
- __init__.py
- __init__.py
- piientity.py
- piienum.py
- extract-block.ndjson
- extract-line.ndjson
- extract-sentence.ndjson
- full-block.ndjson
- full-line.ndjson
- full-sentence.ndjson
- orig.txt
- replace.txt
- tag.txt
- taskfile-error.json
- taskfile.json
- test_file.py
- test_file_taskfile.py
- test_manager.py
- test_manager_add.py
- test_manager_ctx.py
- test_base.py
- test_context.py
- test_norm.py
- test_taskdict.py
- test_bitcoin_address.py
- test_credit_card.py
- test_email.py
- test_ip_address.py
- test_ipn_en.py
- test_abn.py
- test_tfn.py
- test_sin.py
- test_aadhaar.py
- test_ssn.py
- test_ipn_es.py
- test_bank_account.py
- test_govid_es_es.py
- test_govid_es_mx.py
- test_govid_pt_br.py
- test_govid_pt_pt.py
- test_govid_zh_cn.py
- test_misc.py
- .gitignore
- CHANGES.md
- LICENSE
- Makefile
- MANIFEST.in
- README.md
- requirements.txt
- setup.py
- dedup_exact_article.py
- dedup_lines.py
- ram_dedup_lines.py
- requirements.txt
- 01_remove_deplicated_lines.sh
- 02_remove_duplicated_lines_dataset_with_dataset_source.sh
- 03_remove_duplicated_lines_alpha.sh
- 04_remove_duplicated_lines_alpha _memory.sh
- 05_remove_duplicated_lines_alpha __v2_memory.sh
- 06_dedup_exact_examples.sh
- .gitignore
- .gitmodules
- .pre-commit-config.yaml
- __init__.py
- LICENSE
- Makefile
- poetry.lock
- pyproject.toml
- README.md
- requirements.txt
// repository documentation
Was this content helpful?
(0 ratings)
