Spaces:
Sleeping
Sleeping
Add yenniejun/tokenizers-languages files
Browse files- .github/workflows/sync_hf_hub.yaml +20 -0
- README.md +2 -0
- app.py +197 -0
- requirements.txt +64 -0
.github/workflows/sync_hf_hub.yaml
ADDED
|
@@ -0,0 +1,20 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
name: Sync to Hugging Face hub
|
| 2 |
+
on:
|
| 3 |
+
push:
|
| 4 |
+
branches: [main]
|
| 5 |
+
|
| 6 |
+
# to run this workflow manually from the Actions tab
|
| 7 |
+
workflow_dispatch:
|
| 8 |
+
|
| 9 |
+
jobs:
|
| 10 |
+
sync-to-hub:
|
| 11 |
+
runs-on: ubuntu-latest
|
| 12 |
+
steps:
|
| 13 |
+
- uses: actions/checkout@v3
|
| 14 |
+
with:
|
| 15 |
+
fetch-depth: 0
|
| 16 |
+
lfs: true
|
| 17 |
+
- name: Push to hub
|
| 18 |
+
env:
|
| 19 |
+
HF_TOKEN: ${{ secrets.HF_TOKEN }}
|
| 20 |
+
run: git push https://jgalego:$HF_TOKEN@huggingface.co/spaces/jgalego/tokenizers-languages main
|
README.md
CHANGED
|
@@ -12,3 +12,5 @@ short_description: Comparing LLM tokenizers in multiple languages
|
|
| 12 |
---
|
| 13 |
|
| 14 |
Check out the configuration reference at https://huggingface.co/docs/hub/spaces-config-reference
|
|
|
|
|
|
|
|
|
| 12 |
---
|
| 13 |
|
| 14 |
Check out the configuration reference at https://huggingface.co/docs/hub/spaces-config-reference
|
| 15 |
+
|
| 16 |
+
> Adapted from [All languages are NOT created (tokenized) equal](https://www.artfish.ai/p/all-languages-are-not-created-tokenized)
|
app.py
ADDED
|
@@ -0,0 +1,197 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
import streamlit as st
|
| 2 |
+
from collections import defaultdict
|
| 3 |
+
import tqdm
|
| 4 |
+
import transformers
|
| 5 |
+
from transformers import AutoTokenizer
|
| 6 |
+
import pandas as pd
|
| 7 |
+
import matplotlib.pyplot as plt
|
| 8 |
+
import seaborn as sns
|
| 9 |
+
import numpy as np
|
| 10 |
+
import plotly.figure_factory as ff
|
| 11 |
+
import plotly.express as px
|
| 12 |
+
from plotly.subplots import make_subplots
|
| 13 |
+
import plotly.graph_objects as go
|
| 14 |
+
import random, glob
|
| 15 |
+
|
| 16 |
+
@st.cache_data
|
| 17 |
+
def load_data():
|
| 18 |
+
return pd.read_csv('data.csv')
|
| 19 |
+
|
| 20 |
+
def reload_example_text_data():
|
| 21 |
+
random_id = random.choice(val_data['id'])
|
| 22 |
+
tempdf = subset_df[subset_df['id']==random_id]
|
| 23 |
+
tempdf.rename(columns={'lang': 'Language'}, inplace=True)
|
| 24 |
+
tempdf.set_index('Language', inplace=True)
|
| 25 |
+
tempdf = tempdf[['iso', 'text', tokenizer_name]]
|
| 26 |
+
tempdf.columns=['ISO', 'Text', 'Num Tokens']
|
| 27 |
+
tempdf.sort_values(by='ISO', inplace=True)
|
| 28 |
+
st.session_state.examplesdf = tempdf
|
| 29 |
+
|
| 30 |
+
|
| 31 |
+
|
| 32 |
+
|
| 33 |
+
# TODO allow new tokenizers from HF
|
| 34 |
+
tokenizer_names_to_test = [
|
| 35 |
+
"openai/gpt4",
|
| 36 |
+
"Xenova/gpt-4o",
|
| 37 |
+
"Xenova/claude-tokenizer",
|
| 38 |
+
"CohereForAI/aya-101",
|
| 39 |
+
"meta-llama/Meta-Llama-3-70B",
|
| 40 |
+
"mistralai/Mixtral-8x22B-v0.1",
|
| 41 |
+
"google/gemma-7b",
|
| 42 |
+
"facebook/nllb-200-distilled-600M", # Facebook
|
| 43 |
+
"xlm-roberta-base", # old style
|
| 44 |
+
"bert-base-uncased", # old style
|
| 45 |
+
"sentence-transformers/paraphrase-multilingual-MiniLM-L12-v2",
|
| 46 |
+
"bigscience/bloom", # HuggingFace
|
| 47 |
+
"StabilityAI/stablelm-base-alpha-7b", # StableLM with Open Assistant
|
| 48 |
+
"google/flan-t5-base", # Flan T5 (better than T5), Google
|
| 49 |
+
"facebook/mbart-large-50", # Facebook
|
| 50 |
+
"EleutherAI/gpt-neox-20b", # same as Pythia
|
| 51 |
+
]
|
| 52 |
+
|
| 53 |
+
with st.sidebar:
|
| 54 |
+
|
| 55 |
+
st.header('All languages are NOT created (tokenized) equal!')
|
| 56 |
+
link="This project compares the tokenization length for different languages. For some tokenizers, tokenizing a message in one language may result in 10-20x more tokens than a comparable message in another language (e.g. try English vs. Burmese)."
|
| 57 |
+
st.markdown(link)
|
| 58 |
+
link="This is part of a larger project of measuring inequality in NLP. See the original article: [All languages are NOT created (tokenized) equal](https://www.artfish.ai/p/all-languages-are-not-created-tokenized) on [Art Fish Intelligence](https://www.artfish.ai/)."
|
| 59 |
+
st.markdown(link)
|
| 60 |
+
|
| 61 |
+
st.header('Data Visualization')
|
| 62 |
+
st.subheader('Tokenizer')
|
| 63 |
+
# TODO multi-select tokenizers
|
| 64 |
+
tokenizer_name = st.sidebar.selectbox('Select tokenizer', options=tokenizer_names_to_test, label_visibility='collapsed')
|
| 65 |
+
|
| 66 |
+
if tokenizer_name not in ['openai/gpt4']:
|
| 67 |
+
url = f'https://huggingface.co/{tokenizer_name}'
|
| 68 |
+
link = f'Tokenizer is available [on the HuggingFace hub]({url})'
|
| 69 |
+
st.markdown(link, unsafe_allow_html=True)
|
| 70 |
+
else:
|
| 71 |
+
link="Tokenized using [tiktoken](https://github.com/openai/tiktoken)"
|
| 72 |
+
st.markdown(link)
|
| 73 |
+
|
| 74 |
+
|
| 75 |
+
st.subheader('Data')
|
| 76 |
+
with st.spinner('Loading dataset...'):
|
| 77 |
+
val_data = load_data()
|
| 78 |
+
st.success(f'Data loaded: {len(val_data)}')
|
| 79 |
+
|
| 80 |
+
# st.write(val_data.columns, val_data.head())
|
| 81 |
+
|
| 82 |
+
with st.expander('Data Source'):
|
| 83 |
+
st.write("The data in this figure is the validation set of the [Amazon Massive](https://huggingface.co/datasets/AmazonScience/massive/viewer/af-ZA/validation) dataset, which consists of 2033 short sentences and phrases translated into 51 different languages. Learn more about the dataset from [Amazon's blog post](https://www.amazon.science/blog/amazon-releases-51-language-dataset-for-language-understanding)")
|
| 84 |
+
|
| 85 |
+
|
| 86 |
+
st.subheader('Languages')
|
| 87 |
+
languages = st.multiselect(
|
| 88 |
+
'Select languages',
|
| 89 |
+
options=sorted(val_data.lang.unique()),
|
| 90 |
+
default=['English', 'Spanish' ,'Chinese', 'Burmese'],
|
| 91 |
+
max_selections=6,
|
| 92 |
+
label_visibility='collapsed'
|
| 93 |
+
)
|
| 94 |
+
|
| 95 |
+
st.subheader('Figure')
|
| 96 |
+
show_hist = st.checkbox('Show histogram', value=False)
|
| 97 |
+
|
| 98 |
+
|
| 99 |
+
|
| 100 |
+
# dist_marginal = st.radio('Select distribution', options=['box', 'violin', 'rug'], horizontal=True)
|
| 101 |
+
|
| 102 |
+
# with st.spinner('Loading tokenizer...'):
|
| 103 |
+
# tokenizer = AutoTokenizer.from_pretrained(tokenizer_name)
|
| 104 |
+
# st.success(f'Tokenizer loaded: {tokenizer_name}')
|
| 105 |
+
|
| 106 |
+
# # TODO - add the metadata data as well??? later on maybe
|
| 107 |
+
# with st.spinner('Calculating tokenization for data...'):
|
| 108 |
+
# if tokenizer_name not in val_data.columns:
|
| 109 |
+
# val_data[f'{tokenizer_name}'] = val_data.text.apply(lambda x: len(tokenizer.encode(x)))
|
| 110 |
+
# st.success('Completed.')
|
| 111 |
+
|
| 112 |
+
with st.container():
|
| 113 |
+
if tokenizer_name in val_data.columns:
|
| 114 |
+
subset_df = val_data[val_data.lang.isin(languages)]
|
| 115 |
+
subset_data = [val_data[val_data.lang==_lang][tokenizer_name] for _lang in languages]
|
| 116 |
+
|
| 117 |
+
# st.header(f'Comparing languages for {tokenizer_name}')
|
| 118 |
+
|
| 119 |
+
st.subheader(f'Median Token Length for `{tokenizer_name}`')
|
| 120 |
+
metric_cols = st.columns(len(languages))
|
| 121 |
+
for i, _lang in enumerate(languages):
|
| 122 |
+
metric_cols[i].metric(_lang, int(np.median(subset_df[subset_df.lang==_lang][tokenizer_name])))
|
| 123 |
+
|
| 124 |
+
|
| 125 |
+
fig = ff.create_distplot(subset_data, group_labels=languages, show_hist=show_hist)
|
| 126 |
+
|
| 127 |
+
fig.update_layout(
|
| 128 |
+
title=dict(text='Token Distribution', font=dict(size=25), automargin=True, yref='paper', ),
|
| 129 |
+
# title='Distribution of tokens',
|
| 130 |
+
xaxis_title="Number of Tokens",
|
| 131 |
+
yaxis_title="Density",
|
| 132 |
+
height=500
|
| 133 |
+
# title_font_family='"Source Sans Pro", sans-serif'
|
| 134 |
+
)
|
| 135 |
+
st.plotly_chart(fig, use_container_width=True)
|
| 136 |
+
|
| 137 |
+
|
| 138 |
+
# Create figures using px.bar
|
| 139 |
+
shortest = val_data.groupby('lang')[tokenizer_name].median().sort_values().head(7).reset_index()
|
| 140 |
+
shortest["type"] = "shortest"
|
| 141 |
+
longest = val_data.groupby('lang')[tokenizer_name].median().sort_values().tail(7).reset_index()
|
| 142 |
+
longest["type"] = "longest"
|
| 143 |
+
combined = pd.concat([shortest, longest]).reset_index(drop=True).sort_values(by=tokenizer_name, ascending=False)
|
| 144 |
+
color_sequence = px.colors.qualitative.D3 # You can choose other built-in sequences or define your own
|
| 145 |
+
fig = px.bar(combined, x=tokenizer_name, y="lang", orientation='h', color='type', color_discrete_sequence=color_sequence)
|
| 146 |
+
fig.update_traces(hovertemplate='%{y}: %{x} tokens')
|
| 147 |
+
fig.update_layout(
|
| 148 |
+
title=dict(text='Top Langs with Shortest and Longest Median Token Lengths',
|
| 149 |
+
font=dict(size=25), automargin=True, yref='paper', pad=dict(b=20)), # Add more padding below the title
|
| 150 |
+
# title='Distribution of tokens',
|
| 151 |
+
xaxis=dict(
|
| 152 |
+
title="Number of Tokens",
|
| 153 |
+
showgrid=True, # Show vertical gridlines
|
| 154 |
+
gridwidth=1, # Gridline width
|
| 155 |
+
gridcolor='LightGrey' # Gridline color
|
| 156 |
+
),
|
| 157 |
+
yaxis=dict(
|
| 158 |
+
title="",
|
| 159 |
+
),
|
| 160 |
+
height=400,
|
| 161 |
+
showlegend=False # Remove the legend
|
| 162 |
+
)
|
| 163 |
+
st.plotly_chart(fig, use_container_width=True)
|
| 164 |
+
|
| 165 |
+
|
| 166 |
+
|
| 167 |
+
st.subheader('Example Texts')
|
| 168 |
+
reload_example_text_data()
|
| 169 |
+
if st.button("🔄 Randomly sample"):
|
| 170 |
+
reload_example_text_data()
|
| 171 |
+
st.dataframe(st.session_state.examplesdf) # Same as st.write(df)
|
| 172 |
+
|
| 173 |
+
|
| 174 |
+
# val_median_data = val_data.groupby('lang')[tokenizer_name].apply(np.median)
|
| 175 |
+
# val_median_data = val_median_data.sort_values(ascending=False)
|
| 176 |
+
# val_median_data = val_median_data.reset_index()
|
| 177 |
+
# # val_median_data = val_median_data[val_median_data.lang.isin(languages)]
|
| 178 |
+
# val_median_data[tokenizer_name] = val_median_data[tokenizer_name].astype(int)
|
| 179 |
+
# val_median_data.columns = ['Language', 'Median Number of Tokens']
|
| 180 |
+
# # st.write(val_median_data.head())
|
| 181 |
+
# bar_fig = px.bar(
|
| 182 |
+
# val_median_data,
|
| 183 |
+
# y='Language',
|
| 184 |
+
# x='Median Number of Tokens',
|
| 185 |
+
# text_auto='d',
|
| 186 |
+
# orientation='h',
|
| 187 |
+
# hover_data=val_median_data.columns,
|
| 188 |
+
# height=1000,
|
| 189 |
+
# )
|
| 190 |
+
# bar_fig.update_traces(textfont_size=12, textangle=0, cliponaxis=False)
|
| 191 |
+
# bar_fig.update_layout(
|
| 192 |
+
# title=dict(text='Comparison of median token lengths',
|
| 193 |
+
# font=dict(size=20),
|
| 194 |
+
# automargin=True, yref='paper', ),
|
| 195 |
+
# )
|
| 196 |
+
# st.plotly_chart(bar_fig, use_container_width=True)
|
| 197 |
+
|
requirements.txt
ADDED
|
@@ -0,0 +1,64 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
altair==5.3.0
|
| 2 |
+
attrs==23.2.0
|
| 3 |
+
blinker==1.7.0
|
| 4 |
+
cachetools==5.3.3
|
| 5 |
+
certifi==2024.2.2
|
| 6 |
+
charset-normalizer==3.3.2
|
| 7 |
+
click==8.1.7
|
| 8 |
+
contourpy==1.2.0
|
| 9 |
+
cycler==0.12.1
|
| 10 |
+
decorator==5.1.1
|
| 11 |
+
entrypoints==0.4
|
| 12 |
+
filelock==3.13.1
|
| 13 |
+
fonttools==4.49.0
|
| 14 |
+
fsspec==2024.2.0
|
| 15 |
+
gitdb==4.0.11
|
| 16 |
+
GitPython==3.1.42
|
| 17 |
+
huggingface-hub==0.23.0
|
| 18 |
+
idna==3.6
|
| 19 |
+
importlib-metadata==7.0.1
|
| 20 |
+
importlib-resources==6.1.1
|
| 21 |
+
Jinja2==3.1.3
|
| 22 |
+
jsonschema==4.21.1
|
| 23 |
+
kiwisolver==1.4.5
|
| 24 |
+
markdown-it-py==3.0.0
|
| 25 |
+
MarkupSafe==2.1.5
|
| 26 |
+
matplotlib==3.8.3
|
| 27 |
+
mdurl==0.1.2
|
| 28 |
+
numpy==1.26.4
|
| 29 |
+
packaging==23.2
|
| 30 |
+
pandas==2.2.1
|
| 31 |
+
Pillow==10.2.0
|
| 32 |
+
plotly==5.19.0
|
| 33 |
+
protobuf==4.25.3
|
| 34 |
+
pyarrow==15.0.0
|
| 35 |
+
pydeck==0.8.1b0
|
| 36 |
+
Pygments==2.17.2
|
| 37 |
+
Pympler==1.0.1
|
| 38 |
+
pyparsing==3.1.1
|
| 39 |
+
pyrsistent==0.20.0
|
| 40 |
+
python-dateutil==2.8.2
|
| 41 |
+
pytz==2024.1
|
| 42 |
+
pytz-deprecation-shim==0.1.0.post0
|
| 43 |
+
PyYAML==6.0.1
|
| 44 |
+
regex==2023.12.25
|
| 45 |
+
requests==2.31.0
|
| 46 |
+
rich==13.7.1
|
| 47 |
+
scipy==1.12.0
|
| 48 |
+
seaborn==0.13.2
|
| 49 |
+
six==1.16.0
|
| 50 |
+
smmap==5.0.1
|
| 51 |
+
streamlit==1.31.1
|
| 52 |
+
tenacity==8.2.3
|
| 53 |
+
tokenizers==0.15.2
|
| 54 |
+
toml==0.10.2
|
| 55 |
+
toolz==0.12.1
|
| 56 |
+
tornado==6.4
|
| 57 |
+
tqdm==4.66.2
|
| 58 |
+
transformers==4.38.2
|
| 59 |
+
typing_extensions==4.9.0
|
| 60 |
+
tzdata==2024.1
|
| 61 |
+
tzlocal==5.2
|
| 62 |
+
urllib3==2.2.1
|
| 63 |
+
validators==0.22.0
|
| 64 |
+
zipp==3.17.0
|