Spaces:

pauvanbr
/

la-leaderboard-v2

Configuration error

App Files Files Community

pauvanbr commited on May 12

Commit

16cb311

verified ·

1 Parent(s): fcde35f

Upload src/submission/check_validity.py

Browse files

Files changed (1) hide show

src/submission/check_validity.py +129 -0

src/submission/check_validity.py ADDED Viewed

	@@ -0,0 +1,129 @@

+import json
+import logging
+import os
+import re
+from collections import defaultdict
+import huggingface_hub
+from huggingface_hub import ModelCard
+from huggingface_hub.hf_api import ModelInfo, get_safetensors_metadata
+from transformers import AutoConfig, AutoTokenizer
+# TODO: Traducir mensajes de error
+def check_model_card(repo_id: str) -> tuple[bool, str]:
+    """Check whether the model card and license exist and have been filled."""
+    try:
+        card = ModelCard.load(repo_id)
+    except huggingface_hub.utils.EntryNotFoundError:
+        return False, "Please add a model card to your model to explain how you trained/fine-tuned it."
+    # Enforce license metadata
+    if card.data.license is None:
+        if not ("license_name" in card.data and "license_link" in card.data):
+            return False, (
+                "License not found. Please add a license to your model card using the `license` metadata or a"
+                " `license_name`/`license_link` pair."
+            )
+    # Enforce card content
+    if len(card.text) < 200:
+        return False, "Please add a description to your model card, it is too short."
+    return True, ""
+def is_model_on_hub(
+    model_name: str, revision: str, token: str = None, trust_remote_code=False, test_tokenizer=False
+) -> tuple[bool, str]:
+    """Check whether the model model_name is on the hub, and whether it (and its tokenizer) can be loaded with AutoClasses."""
+    try:
+        config = AutoConfig.from_pretrained(
+            model_name, revision=revision, trust_remote_code=trust_remote_code, token=token, force_download=True
+        )
+        if test_tokenizer:
+            try:
+                AutoTokenizer.from_pretrained(
+                    model_name, revision=revision, trust_remote_code=trust_remote_code, token=token
+                )
+            except ValueError as e:
+                return (False, f"uses a tokenizer which is not in a transformers release: {e}", None)
+            except Exception as e:
+                return (
+                    False,
+                    f"'s tokenizer cannot be loaded. Is your tokenizer class in a stable transformers release, and correctly configured? {e}",
+                    None,
+                )
+        return True, None, config
+    except ValueError:
+        return (
+            False,
+            "needs to be launched with `trust_remote_code=True`. For safety reason, we do not allow these models to be automatically submitted to the leaderboard.",
+            None,
+        )
+    except Exception as e:
+        if "You are trying to access a gated repo." in str(e):
+            return True, "uses a gated model.", None
+        return False, f"was not found or misconfigured on the hub! Error raised was {e.args[0]}", None
+def get_model_size(model_info: ModelInfo, precision: str) -> float:
+    size_pattern = re.compile(r"(\d+\.)?\d+(b|m)")
+    safetensors = None
+    try:
+        safetensors = get_safetensors_metadata(model_info.id)
+    except Exception as e:
+        logging.error(f"Failed to get safetensors metadata for model {model_info.id}: {str(e)}")
+    if safetensors is not None:
+        model_size = round(sum(safetensors.parameter_count.values()) / 1e9, 3)
+    else:
+        try:
+            size_match = re.search(size_pattern, model_info.id.lower())
+            if size_match:
+                model_size = size_match.group(0)
+                model_size = round(
+                    float(model_size[:-1]) if model_size[-1] == "b" else float(model_size[:-1]) / 1e3, 3
+                )
+            else:
+                return -1  # Unknown model size
+        except AttributeError:
+            logging.warning(f"Unable to parse model size from ID: {model_info.id}")
+            return -1  # Unknown model size
+    size_factor = 8 if (precision == "GPTQ" or "gptq" in model_info.id.lower()) else 1
+    model_size = size_factor * model_size
+    return model_size
+def get_model_arch(model_info: ModelInfo):
+    """Get the model architecture from the configuration."""
+    return model_info.config.get("architectures", "Unknown")
+def already_submitted_models(requested_models_dir: str) -> set[str]:
+    """Gather a list of already submitted models to avoid duplicates."""
+    depth = 1
+    file_names = []
+    users_to_submission_dates = defaultdict(list)
+    for root, _, files in os.walk(requested_models_dir):
+        current_depth = root.count(os.sep) - requested_models_dir.count(os.sep)
+        if current_depth == depth:
+            for file in files:
+                if not file.endswith(".json"):
+                    continue
+                with open(os.path.join(root, file), "r") as f:
+                    info = json.load(f)
+                    file_names.append(f"{info['model']}_{info['revision']}_{info['precision']}")
+                    # Select organisation
+                    if info["model"].count("/") == 0 or "submitted_time" not in info:
+                        continue
+                    organisation, _ = info["model"].split("/")
+                    users_to_submission_dates[organisation].append(info["submitted_time"])
+    return set(file_names), users_to_submission_dates