Instructions to use OpenFormosa/PangolinTokenizer with libraries, inference providers, notebooks, and local apps. Follow these links to get started.
- Libraries
- Transformers
How to use OpenFormosa/PangolinTokenizer with Transformers:
# Load model directly from transformers import AutoModel model = AutoModel.from_pretrained("OpenFormosa/PangolinTokenizer", device_map="auto") - Notebooks
- Google Colab
- Kaggle
Download evaluation_report.json from OpenFormosa/PangolinTokenizer: direct link, hf CLI and curl.
- Browser
- Download file 1.63 kB
-
https://huggingface.co/OpenFormosa/PangolinTokenizer/resolve/82e506e0a6574454c826e805c269326120571f63/evaluation_report.json
- Command line
-
hf download hf://OpenFormosa/PangolinTokenizer@82e506e0a6574454c826e805c269326120571f63/evaluation_report.json
-
curl -L -o evaluation_report.json https://huggingface.co/OpenFormosa/PangolinTokenizer/resolve/82e506e0a6574454c826e805c269326120571f63/evaluation_report.json
1.63 kB
| { | |
| "ok": true, | |
| "failures": [], | |
| "sections": { | |
| "Timestamp / rich transcription checks": { | |
| "rich_transcription_token_ids": { | |
| "<|transcript_start|>": 114670, | |
| "<|transcript_end|>": 114671, | |
| "<|segment_start|>": 114672, | |
| "<|segment_end|>": 114673, | |
| "<|speaker|>": 114674, | |
| "<|start_time|>": 114675, | |
| "<|end_time|>": 114676, | |
| "<|duration|>": 114677, | |
| "<|content|>": 114678, | |
| "<|non_speech_event|>": 114679 | |
| }, | |
| "timestamp_precision_digits": 2, | |
| "json_roundtrip_ok": true, | |
| "dense_timestamp_tokens_found": [], | |
| "missing_rich_transcription_tokens": [], | |
| "timestamp_strings_present": { | |
| "0.00": true, | |
| "3.42": true, | |
| "10.25": true, | |
| "3575.50": true | |
| }, | |
| "non_speech_labels_present": { | |
| "[Silence]": true, | |
| "[Noise]": true, | |
| "[Music]": true, | |
| "[Unintelligible Speech]": true | |
| }, | |
| "text_roundtrip_ok": { | |
| "traditional_chinese": true, | |
| "bopomofo_mixed_romanization": true, | |
| "json_syntax": true | |
| }, | |
| "required_fields_ok": true, | |
| "parse_error": null | |
| } | |
| }, | |
| "rich_transcription_token_ids": { | |
| "<|transcript_start|>": 114670, | |
| "<|transcript_end|>": 114671, | |
| "<|segment_start|>": 114672, | |
| "<|segment_end|>": 114673, | |
| "<|speaker|>": 114674, | |
| "<|start_time|>": 114675, | |
| "<|end_time|>": 114676, | |
| "<|duration|>": 114677, | |
| "<|content|>": 114678, | |
| "<|non_speech_event|>": 114679 | |
| }, | |
| "timestamp_precision_digits": 2, | |
| "json_roundtrip_ok": true, | |
| "dense_timestamp_tokens_found": [] | |
| } | |