Spaces:

evijit
/

ModelVerse

Running

App Files Files Community

Avijit Ghosh commited on 27 days ago

Commit

fd50825

1 Parent(s): ce43639

fixed bugs

Browse files

Files changed (3) hide show

README.md +1 -1
app.py +102 -61
requirements.txt +1 -1

README.md CHANGED Viewed

@@ -4,7 +4,7 @@ emoji: 🚀
 colorFrom: blue
 colorTo: green
 sdk: gradio
-sdk_version: 5.40.0
 app_file: app.py
 pinned: false
 license: apache-2.0

 colorFrom: blue
 colorTo: green
 sdk: gradio
+sdk_version: 6.1.0
 app_file: app.py
 pinned: false
 license: apache-2.0

app.py CHANGED Viewed

@@ -2,7 +2,8 @@ import gradio as gr
 import pandas as pd
 import plotly.express as px
 import time
-from datasets import load_dataset
 # Using the stable, community-built RangeSlider component
 from gradio_rangeslider import RangeSlider
 import datetime # Import the datetime module
@@ -19,25 +20,22 @@ PIPELINE_TAGS = [ 'text-generation', 'text-to-image', 'text-classification', 'te
 def load_models_data():
     overall_start_time = time.time()
-    print(f"Attempting to load dataset from Hugging Face Hub: {HF_DATASET_ID}")
     try:
-        dataset_dict = load_dataset(HF_DATASET_ID)
-        df = dataset_dict[list(dataset_dict.keys())[0]].to_pandas()
-        if 'params' in df.columns:
-            df['params'] = pd.to_numeric(df['params'], errors='coerce').fillna(-1)
-        else:
-            df['params'] = -1
-        if 'createdAt' in df.columns:
-            df['createdAt'] = pd.to_datetime(df['createdAt'], errors='coerce')
-        msg = f"Successfully loaded dataset in {time.time() - overall_start_time:.2f}s."
         print(msg)
-        return df, True, msg
     except Exception as e:
-        err_msg = f"Failed to load dataset. Error: {e}"
         print(err_msg)
-        return pd.DataFrame(), False, err_msg
 def get_param_range_values(param_range_labels):
     min_label, max_label = param_range_labels
@@ -45,44 +43,76 @@ def get_param_range_values(param_range_labels):
     max_val = float('inf') if '>' in max_label else float(max_label.replace('B', ''))
     return min_val, max_val
-def make_treemap_data(df, count_by, top_k=25, tag_filter=None, pipeline_filter=None, param_range=None, skip_orgs=None, include_unknown_param_size=True, created_after_date: float = None):
-    if df is None or df.empty: return pd.DataFrame()
-    filtered_df = df.copy()
-    if not include_unknown_param_size and 'params' in filtered_df.columns:
-        filtered_df = filtered_df[filtered_df['params'] != -1]
     col_map = { "Audio & Speech": "is_audio_speech", "Music": "has_music", "Robotics": "has_robot", "Biomedical": "is_biomed", "Time series": "has_series", "Sciences": "has_science", "Video": "has_video", "Images": "has_image", "Text": "has_text" }
-    if tag_filter and tag_filter in col_map and col_map[tag_filter] in filtered_df.columns:
-        filtered_df = filtered_df[filtered_df[col_map[tag_filter]]]
-    if pipeline_filter and "pipeline_tag" in filtered_df.columns:
-        filtered_df = filtered_df[filtered_df["pipeline_tag"].astype(str) == pipeline_filter]
     if param_range:
         min_params, max_params = get_param_range_values(param_range)
         is_default_range = (param_range[0] == PARAM_CHOICES[0] and param_range[1] == PARAM_CHOICES[-1])
-        if not is_default_range and 'params' in filtered_df.columns:
-            if min_params is not None: filtered_df = filtered_df[filtered_df['params'] >= min_params]
-            if max_params is not None and max_params != float('inf'): filtered_df = filtered_df[filtered_df['params'] < max_params]
-    # --- CORRECTED DATE FILTER LOGIC FOR FLOAT TIMESTAMP ---
-    if created_after_date is not None and 'createdAt' in filtered_df.columns:
-        # Drop rows where 'createdAt' could not be parsed to avoid errors
-        filtered_df = filtered_df.dropna(subset=['createdAt'])
-        # Convert the Unix timestamp (float) from the UI into a Python date object
-        filter_date = datetime.datetime.fromtimestamp(created_after_date).date()
-        # Compare its date part with the date part of the 'createdAt' column.
-        filtered_df = filtered_df[filtered_df['createdAt'].dt.date > filter_date]
-    if skip_orgs and len(skip_orgs) > 0 and "organization" in filtered_df.columns:
-        filtered_df = filtered_df[~filtered_df["organization"].isin(skip_orgs)]
-    if filtered_df.empty: return pd.DataFrame()
-    if count_by not in filtered_df.columns: filtered_df[count_by] = 0.0
-    filtered_df[count_by] = pd.to_numeric(filtered_df[count_by], errors='coerce').fillna(0.0)
-    org_totals = filtered_df.groupby("organization")[count_by].sum().nlargest(top_k, keep='first')
-    top_orgs_list = org_totals.index.tolist()
-    treemap_data = filtered_df[filtered_df["organization"].isin(top_orgs_list)][["id", "organization", count_by]].copy()
     treemap_data["root"] = "models"
     return treemap_data
@@ -109,7 +139,7 @@ custom_css = """
 """
 with gr.Blocks(title="🤗 ModelVerse Explorer", fill_width=True, css=custom_css) as demo:
-    models_data_state = gr.State(pd.DataFrame())
     loading_complete_state = gr.State(False)
     with gr.Row():
@@ -166,19 +196,30 @@ with gr.Blocks(title="🤗 ModelVerse Explorer", fill_width=True, css=custom_css
                                filter_choice_radio, [tag_filter_dropdown, pipeline_filter_dropdown])
     def load_and_generate_initial_plot(progress=gr.Progress()):
-        progress(0, desc=f"Loading dataset '{HF_DATASET_ID}'...")
-        current_df, load_success_flag, status_msg_from_load = pd.DataFrame(), False, ""
         try:
-            current_df, load_success_flag, status_msg_from_load = load_models_data()
             if load_success_flag:
-                progress(0.5, desc="Processing data...")
-                ts = pd.to_datetime(current_df['data_download_timestamp'].iloc[0], utc=True) if 'data_download_timestamp' in current_df.columns and pd.notna(current_df['data_download_timestamp'].iloc[0]) else None
                 date_display = ts.strftime('%B %d, %Y, %H:%M:%S %Z') if ts else "Pre-processed (date unavailable)"
-                param_count = (current_df['params'] != -1).sum()
                 data_info_text = (f"### Data Information\n- Source: `{HF_DATASET_ID}`\n- Status: {status_msg_from_load}\n"
-                                  f"- Total models loaded: {len(current_df):,}\n- Models with known parameter counts: {param_count:,}\n"
-                                  f"- Models with unknown parameter counts: {len(current_df) - param_count:,}\n- Data as of: {date_display}\n")
             else:
                 data_info_text = f"### Data Load Failed\n- {status_msg_from_load}"
         except Exception as e:
@@ -189,21 +230,21 @@ with gr.Blocks(title="🤗 ModelVerse Explorer", fill_width=True, css=custom_css
         progress(0.6, desc="Generating initial plot...")
         initial_plot, initial_status = ui_generate_plot_controller(
             "downloads", "None", None, None, PARAM_CHOICES_DEFAULT_INDICES, 25,
-            "TheBloke,MaziyarPanahi,unsloth,modularai,Gensyn,bartowski", True, None, current_df, progress
         )
-        return current_df, load_success_flag, data_info_text, initial_status, initial_plot
     def ui_generate_plot_controller(metric_choice, filter_type, tag_choice, pipeline_choice,
                                    param_range_indices, k_orgs, skip_orgs_input, include_unknown_param_size_flag,
-                                   created_after_date, df_current_models, progress=gr.Progress()):
-        if df_current_models.empty:
             return create_treemap(pd.DataFrame(), metric_choice, "Error: Model Data Not Loaded"), "Model data is not loaded."
         progress(0.1, desc="Preparing data...")
         param_labels = [PARAM_CHOICES[int(param_range_indices[0])], PARAM_CHOICES[int(param_range_indices[1])]]
         treemap_df = make_treemap_data(
-            df_current_models, metric_choice, k_orgs,
             tag_choice if filter_type == "Tag Filter" else None,
             pipeline_choice if filter_type == "Pipeline Filter" else None,
             param_labels, [org.strip() for org in skip_orgs_input.split(',') if org.strip()],

 import pandas as pd
 import plotly.express as px
 import time
+import duckdb
+from huggingface_hub import list_repo_files
 # Using the stable, community-built RangeSlider component
 from gradio_rangeslider import RangeSlider
 import datetime # Import the datetime module
 def load_models_data():
     overall_start_time = time.time()
+    print(f"Attempting to load dataset metadata from Hugging Face Hub: {HF_DATASET_ID}")
     try:
+        files = list_repo_files(HF_DATASET_ID, repo_type="dataset")
+        parquet_files = [f for f in files if f.endswith('.parquet')]
+        if not parquet_files:
+            return [], False, "No parquet files found in dataset."
+        urls = [f"https://huggingface.co/datasets/{HF_DATASET_ID}/resolve/main/{f}" for f in parquet_files]
+        msg = f"Successfully identified {len(urls)} parquet files in {time.time() - overall_start_time:.2f}s."
         print(msg)
+        return urls, True, msg
     except Exception as e:
+        err_msg = f"Failed to load dataset metadata. Error: {e}"
         print(err_msg)
+        return [], False, err_msg
 def get_param_range_values(param_range_labels):
     min_label, max_label = param_range_labels
     max_val = float('inf') if '>' in max_label else float(max_label.replace('B', ''))
     return min_val, max_val
+def make_treemap_data(parquet_urls, count_by, top_k=25, tag_filter=None, pipeline_filter=None, param_range=None, skip_orgs=None, include_unknown_param_size=True, created_after_date: float = None):
+    if not parquet_urls: return pd.DataFrame()
+    con = duckdb.connect()
+    con.execute("INSTALL httpfs; LOAD httpfs;")
+    urls_str = ", ".join([f"'{u}'" for u in parquet_urls])
+    con.execute(f"CREATE VIEW models AS SELECT * FROM read_parquet([{urls_str}])")
+    where_clauses = []
+    if not include_unknown_param_size:
+        where_clauses.append("params IS NOT NULL AND params != -1")
     col_map = { "Audio & Speech": "is_audio_speech", "Music": "has_music", "Robotics": "has_robot", "Biomedical": "is_biomed", "Time series": "has_series", "Sciences": "has_science", "Video": "has_video", "Images": "has_image", "Text": "has_text" }
+    if tag_filter and tag_filter in col_map:
+        where_clauses.append(f"{col_map[tag_filter]} = true")
+    if pipeline_filter:
+        where_clauses.append(f"pipeline_tag = '{pipeline_filter}'")
     if param_range:
         min_params, max_params = get_param_range_values(param_range)
         is_default_range = (param_range[0] == PARAM_CHOICES[0] and param_range[1] == PARAM_CHOICES[-1])
+        if not is_default_range:
+            conditions = []
+            if min_params is not None:
+                conditions.append(f"params >= {min_params}")
+            if max_params is not None and max_params != float('inf'):
+                conditions.append(f"params < {max_params}")
+            if conditions:
+                where_clauses.append("(" + " AND ".join(conditions) + ")")
+    if created_after_date is not None:
+        where_clauses.append(f"CAST(createdAt AS TIMESTAMPTZ) > to_timestamp({created_after_date})")
+    if skip_orgs and len(skip_orgs) > 0:
+        orgs_str = ", ".join([f"'{o}'" for o in skip_orgs])
+        where_clauses.append(f"organization NOT IN ({orgs_str})")
+    where_sql = " WHERE " + " AND ".join(where_clauses) if where_clauses else ""
+    metric = f"COALESCE({count_by}, 0)"
+    query = f"""
+        SELECT organization, SUM({metric}) as total_metric
+        FROM models
+        {where_sql}
+        GROUP BY organization
+        ORDER BY total_metric DESC
+        LIMIT {top_k}
+    """
+    top_orgs_df = con.execute(query).df()
+    if top_orgs_df.empty:
+        return pd.DataFrame()
+    top_orgs_list = top_orgs_df['organization'].tolist()
+    orgs_filter = ", ".join([f"'{o}'" for o in top_orgs_list])
+    detail_query = f"""
+        SELECT id, organization, {metric} as {count_by}
+        FROM models
+        {where_sql}
+        AND organization IN ({orgs_filter})
+    """
+    treemap_data = con.execute(detail_query).df()
     treemap_data["root"] = "models"
     return treemap_data
 """
 with gr.Blocks(title="🤗 ModelVerse Explorer", fill_width=True, css=custom_css) as demo:
+    models_data_state = gr.State([])
     loading_complete_state = gr.State(False)
     with gr.Row():
                                filter_choice_radio, [tag_filter_dropdown, pipeline_filter_dropdown])
     def load_and_generate_initial_plot(progress=gr.Progress()):
+        progress(0, desc=f"Loading dataset metadata '{HF_DATASET_ID}'...")
+        parquet_urls, load_success_flag, status_msg_from_load = [], False, ""
         try:
+            parquet_urls, load_success_flag, status_msg_from_load = load_models_data()
             if load_success_flag:
+                progress(0.5, desc="Processing metadata...")
+                # Quick query to get stats
+                con = duckdb.connect()
+                con.execute("INSTALL httpfs; LOAD httpfs;")
+                urls_str = ", ".join([f"'{u}'" for u in parquet_urls])
+                con.execute(f"CREATE VIEW models AS SELECT * FROM read_parquet([{urls_str}])")
+                # Get total count and timestamp
+                stats = con.execute("SELECT count(*), max(data_download_timestamp), count(params) FROM models").fetchone()
+                total_count = stats[0]
+                ts = stats[1] # Timestamp object
+                param_count = stats[2]
                 date_display = ts.strftime('%B %d, %Y, %H:%M:%S %Z') if ts else "Pre-processed (date unavailable)"
                 data_info_text = (f"### Data Information\n- Source: `{HF_DATASET_ID}`\n- Status: {status_msg_from_load}\n"
+                                  f"- Total models loaded: {total_count:,}\n- Models with known parameter counts: {param_count:,}\n"
+                                  f"- Models with unknown parameter counts: {total_count - param_count:,}\n- Data as of: {date_display}\n")
             else:
                 data_info_text = f"### Data Load Failed\n- {status_msg_from_load}"
         except Exception as e:
         progress(0.6, desc="Generating initial plot...")
         initial_plot, initial_status = ui_generate_plot_controller(
             "downloads", "None", None, None, PARAM_CHOICES_DEFAULT_INDICES, 25,
+            "TheBloke,MaziyarPanahi,unsloth,modularai,Gensyn,bartowski", True, None, parquet_urls, progress
         )
+        return parquet_urls, load_success_flag, data_info_text, initial_status, initial_plot
     def ui_generate_plot_controller(metric_choice, filter_type, tag_choice, pipeline_choice,
                                    param_range_indices, k_orgs, skip_orgs_input, include_unknown_param_size_flag,
+                                   created_after_date, parquet_urls, progress=gr.Progress()):
+        if not parquet_urls:
             return create_treemap(pd.DataFrame(), metric_choice, "Error: Model Data Not Loaded"), "Model data is not loaded."
         progress(0.1, desc="Preparing data...")
         param_labels = [PARAM_CHOICES[int(param_range_indices[0])], PARAM_CHOICES[int(param_range_indices[1])]]
         treemap_df = make_treemap_data(
+            parquet_urls, metric_choice, k_orgs,
             tag_choice if filter_type == "Tag Filter" else None,
             pipeline_choice if filter_type == "Pipeline Filter" else None,
             param_labels, [org.strip() for org in skip_orgs_input.split(',') if org.strip()],

requirements.txt CHANGED Viewed

@@ -1,3 +1,3 @@
 plotly
 duckdb
-gradio-rangeslider

 plotly
 duckdb
+gradio-rangeslider