gguf-my-repo

Runtime error

App Files Files Community

Ffftdtd5dtft commited on Aug 29, 2024

Commit

5532bbf

verified ·

1 Parent(s): 0d6d15f

Update app.py

Browse files

Files changed (1) hide show

app.py +129 -32

app.py CHANGED Viewed

@@ -17,20 +17,24 @@ HF_TOKEN = os.environ.get("HF_TOKEN")
 def generate_importance_matrix(model_path, train_data_path):
     imatrix_command = f"./llama-imatrix -m ../{model_path} -f {train_data_path} -ngl 99 --output-frequency 10"
     os.chdir("llama.cpp")
     if not os.path.isfile(f"../{model_path}"):
         raise Exception(f"Model file not found: {model_path}")
     process = subprocess.Popen(imatrix_command, shell=True)
     try:
-        process.wait(timeout=60)
     except subprocess.TimeoutExpired:
-        print("Imatrix computation timed out. Sending SIGINT...")
         process.send_signal(signal.SIGINT)
         try:
-            process.wait(timeout=5)
         except subprocess.TimeoutExpired:
-            print("Imatrix proc still didn't term. Forecfully terming...")
             process.kill()
     os.chdir("..")
 def split_upload_model(model_path, repo_id, oauth_token: gr.OAuthToken | None, split_max_tensors=256, split_max_size=None):
     if oauth_token.token is None:
@@ -39,14 +43,20 @@ def split_upload_model(model_path, repo_id, oauth_token: gr.OAuthToken | None, s
     if split_max_size:
         split_cmd += f" --split-max-size {split_max_size}"
     split_cmd += f" {model_path} {model_path.split('.')[0]}"
     result = subprocess.run(split_cmd, shell=True, capture_output=True, text=True)
     if result.returncode != 0:
         raise Exception(f"Error splitting the model: {result.stderr}")
     sharded_model_files = [f for f in os.listdir('.') if f.startswith(model_path.split('.')[0])]
     if sharded_model_files:
         api = HfApi(token=oauth_token.token)
         for file in sharded_model_files:
             file_path = os.path.join('.', file)
             try:
                 api.upload_file(
                     path_or_fileobj=file_path,
@@ -57,6 +67,7 @@ def split_upload_model(model_path, repo_id, oauth_token: gr.OAuthToken | None, s
                 raise Exception(f"Error uploading file {file_path}: {e}")
     else:
         raise Exception("No sharded files found.")
 def process_model(model_id, q_method, use_imatrix, imatrix_q_method, private_repo, train_data_file, split_model, split_max_tensors, split_max_size, oauth_token: gr.OAuthToken | None):
     if oauth_token.token is None:
@@ -66,29 +77,43 @@ def process_model(model_id, q_method, use_imatrix, imatrix_q_method, private_rep
     try:
         api = HfApi(token=oauth_token.token)
-        api.snapshot_download(repo_id=model_id, local_dir=model_name, local_dir_use_symlinks=False)
-        all_files = []
-        for root, _, files in os.walk(model_name):
-            for file in files:
-                all_files.append(os.path.join(root, file))
-        if not all_files:
-            raise FileNotFoundError("No files found in the downloaded model directory.")
-        for file_path in all_files:
-            try:
-                gguf_model_file = f"{os.path.splitext(file_path)[0]}.gguf"
-                conversion_command = f"python llama.cpp/convert_hf_to_gguf.py {file_path} --outfile {gguf_model_file}"
-                result = subprocess.run(conversion_command, shell=True, capture_output=True)
-                if result.returncode == 0:
-                    model_file = gguf_model_file
-                    break
-            except Exception as e:
-                print(f"Conversion attempt failed for {file_path}: {e}")
         if model_file is None:
-            raise Exception("Unable to find or convert a suitable model file to GGUF format.")
         imatrix_path = "llama.cpp/imatrix.dat"
@@ -96,12 +121,16 @@ def process_model(model_id, q_method, use_imatrix, imatrix_q_method, private_rep
             if train_data_file:
                 train_data_path = train_data_file.name
             else:
-                train_data_path = "groups_merged.txt"
             if not os.path.isfile(train_data_path):
                 raise Exception(f"Training data file not found: {train_data_path}")
-            generate_importance_matrix(model_file, train_data_path)
         username = whoami(oauth_token.token)["name"]
         quantized_gguf_name = f"{model_name.lower()}-{imatrix_q_method.lower()}-imat.gguf" if use_imatrix else f"{model_name.lower()}-{q_method.lower()}.gguf"
@@ -109,22 +138,81 @@ def process_model(model_id, q_method, use_imatrix, imatrix_q_method, private_rep
         os.chdir("llama.cpp")
         if use_imatrix:
-            quantise_ggml = f"./llama-quantize --imatrix {imatrix_path} ../{model_file} ../{quantized_gguf_path} {imatrix_q_method}"
         else:
-            quantise_ggml = f"./llama-quantize ../{model_file} ../{quantized_gguf_path} {q_method}"
         result = subprocess.run(quantise_ggml, shell=True, capture_output=True)
         os.chdir("..")
         if result.returncode != 0:
             raise Exception(f"Error quantizing: {result.stderr}")
         new_repo_url = api.create_repo(repo_id=f"{username}/{model_name}-{imatrix_q_method if use_imatrix else q_method}-GGUF", exist_ok=True, private=private_repo)
         new_repo_id = new_repo_url.repo_id
         if split_model:
             split_upload_model(quantized_gguf_path, new_repo_id, oauth_token, split_max_tensors, split_max_size)
         else:
             try:
                 api.upload_file(
                     path_or_fileobj=quantized_gguf_path,
                     path_in_repo=quantized_gguf_name,
@@ -135,6 +223,7 @@ def process_model(model_id, q_method, use_imatrix, imatrix_q_method, private_rep
         if use_imatrix and os.path.isfile(imatrix_path):
             try:
                 api.upload_file(
                     path_or_fileobj=imatrix_path,
                     path_in_repo="imatrix.dat",
@@ -143,6 +232,13 @@ def process_model(model_id, q_method, use_imatrix, imatrix_q_method, private_rep
             except Exception as e:
                 raise Exception(f"Error uploading imatrix.dat: {e}")
         return (
             f'Find your repo <a href=\'{new_repo_url}\' target="_blank" style="text-decoration:underline">here</a>',
             "llama.png",
@@ -151,11 +247,12 @@ def process_model(model_id, q_method, use_imatrix, imatrix_q_method, private_rep
         return (f"Error: {e}", "error.png")
     finally:
         shutil.rmtree(model_name, ignore_errors=True)
 css="""/* Custom CSS to allow scrolling */
 .gradio-container {overflow-y: auto;}
 """
-with gr.Blocks(css=css) as demo:
     gr.Markdown("You must be logged in to use GGUF-my-repo.")
     gr.LoginButton(min_width=250)
@@ -178,7 +275,7 @@ with gr.Blocks(css=css) as demo:
         ["IQ3_M", "IQ3_XXS", "Q4_K_M", "Q4_K_S", "IQ4_NL", "IQ4_XS", "Q5_K_M", "Q5_K_S"],
         label="Imatrix Quantization Method",
         info="GGML imatrix quants type",
-        value="IQ4_NL",
         filterable=False,
         visible=False
     )
@@ -222,7 +319,7 @@ with gr.Blocks(css=css) as demo:
     def update_visibility(use_imatrix):
         return gr.update(visible=not use_imatrix), gr.update(visible=use_imatrix), gr.update(visible=use_imatrix)
     use_imatrix.change(
         fn=update_visibility,
         inputs=use_imatrix,
@@ -261,7 +358,7 @@ with gr.Blocks(css=css) as demo:
     )
 def restart_space():
-    HfApi().restart_space(repo_id="ggml-org/gguf-my-repo", token=HF_TOKEN, factory_reboot=True)
 scheduler = BackgroundScheduler()
 scheduler.add_job(restart_space, "interval", seconds=21600)

 def generate_importance_matrix(model_path, train_data_path):
     imatrix_command = f"./llama-imatrix -m ../{model_path} -f {train_data_path} -ngl 99 --output-frequency 10"
     os.chdir("llama.cpp")
+    print(f"Current working directory: {os.getcwd()}")
+    print(f"Files in the current directory: {os.listdir('.')}")
     if not os.path.isfile(f"../{model_path}"):
         raise Exception(f"Model file not found: {model_path}")
+    print("Running imatrix command...")
     process = subprocess.Popen(imatrix_command, shell=True)
     try:
+        process.wait(timeout=60)
     except subprocess.TimeoutExpired:
+        print("Imatrix computation timed out. Sending SIGINT to allow graceful termination...")
         process.send_signal(signal.SIGINT)
         try:
+            process.wait(timeout=5)
         except subprocess.TimeoutExpired:
+            print("Imatrix proc still didn't term. Forecfully terming process...")
             process.kill()
     os.chdir("..")
+    print("Importance matrix generation completed.")
 def split_upload_model(model_path, repo_id, oauth_token: gr.OAuthToken | None, split_max_tensors=256, split_max_size=None):
     if oauth_token.token is None:
     if split_max_size:
         split_cmd += f" --split-max-size {split_max_size}"
     split_cmd += f" {model_path} {model_path.split('.')[0]}"
+    print(f"Split command: {split_cmd}")
     result = subprocess.run(split_cmd, shell=True, capture_output=True, text=True)
+    print(f"Split command stdout: {result.stdout}")
+    print(f"Split command stderr: {result.stderr}")
     if result.returncode != 0:
         raise Exception(f"Error splitting the model: {result.stderr}")
+    print("Model split successfully!")
     sharded_model_files = [f for f in os.listdir('.') if f.startswith(model_path.split('.')[0])]
     if sharded_model_files:
+        print(f"Sharded model files: {sharded_model_files}")
         api = HfApi(token=oauth_token.token)
         for file in sharded_model_files:
             file_path = os.path.join('.', file)
+            print(f"Uploading file: {file_path}")
             try:
                 api.upload_file(
                     path_or_fileobj=file_path,
                 raise Exception(f"Error uploading file {file_path}: {e}")
     else:
         raise Exception("No sharded files found.")
+    print("Sharded model has been uploaded successfully!")
 def process_model(model_id, q_method, use_imatrix, imatrix_q_method, private_repo, train_data_file, split_model, split_max_tensors, split_max_size, oauth_token: gr.OAuthToken | None):
     if oauth_token.token is None:
     try:
         api = HfApi(token=oauth_token.token)
+        # Download only necessary files based on model format
+        dl_pattern = ["*.md", "*.json"]
+        pattern = (
+            "*.safetensors"
+            if any(
+                file.path.endswith(".safetensors")
+                for file in api.list_repo_tree(
+                    repo_id=model_id,
+                    recursive=True,
+                )
+            )
+            else "*.bin"
+        )
+        dl_pattern += pattern
+        api.snapshot_download(repo_id=model_id, local_dir=model_name, local_dir_use_symlinks=False, allow_patterns=dl_pattern)
+        print("Model downloaded successfully!")
+        print(f"Current working directory: {os.getcwd()}")
+        print(f"Model directory contents: {os.listdir(model_name)}")
+        # Find downloaded model file
+        for filename in os.listdir(model_name):
+            if filename.endswith((".bin", ".safetensors")):
+                model_file = os.path.join(model_name, filename)
+                break
         if model_file is None:
+            raise FileNotFoundError("No model file found in the downloaded files.")
+        # Convert to GGUF
+        gguf_model_file = f"{os.path.splitext(model_file)[0]}.gguf"
+        conversion_command = f"python llama.cpp/convert_hf_to_gguf.py {model_file} --outfile {gguf_model_file}"
+        result = subprocess.run(conversion_command, shell=True, capture_output=True)
+        if result.returncode != 0:
+            raise Exception(f"Error converting to GGUF: {result.stderr}")
+        print("Model converted to GGUF successfully!")
+        print(f"Converted model path: {gguf_model_file}")
         imatrix_path = "llama.cpp/imatrix.dat"
             if train_data_file:
                 train_data_path = train_data_file.name
             else:
+                train_data_path = "groups_merged.txt" #fallback calibration dataset
+            print(f"Training data file path: {train_data_path}")
             if not os.path.isfile(train_data_path):
                 raise Exception(f"Training data file not found: {train_data_path}")
+            generate_importance_matrix(gguf_model_file, train_data_path)
+        else:
+            print("Not using imatrix quantization.")
         username = whoami(oauth_token.token)["name"]
         quantized_gguf_name = f"{model_name.lower()}-{imatrix_q_method.lower()}-imat.gguf" if use_imatrix else f"{model_name.lower()}-{q_method.lower()}.gguf"
         os.chdir("llama.cpp")
         if use_imatrix:
+            quantise_ggml = f"./llama-quantize --imatrix {imatrix_path} ../{gguf_model_file} ../{quantized_gguf_path} {imatrix_q_method}"
         else:
+            quantise_ggml = f"./llama-quantize ../{gguf_model_file} ../{quantized_gguf_path} {q_method}"
         result = subprocess.run(quantise_ggml, shell=True, capture_output=True)
         os.chdir("..")
         if result.returncode != 0:
             raise Exception(f"Error quantizing: {result.stderr}")
+        print(f"Quantized successfully with {imatrix_q_method if use_imatrix else q_method} option!")
+        print(f"Quantized model path: {quantized_gguf_path}")
         new_repo_url = api.create_repo(repo_id=f"{username}/{model_name}-{imatrix_q_method if use_imatrix else q_method}-GGUF", exist_ok=True, private=private_repo)
         new_repo_id = new_repo_url.repo_id
+        print("Repo created successfully!", new_repo_url)
+        try:
+            card = ModelCard.load(model_id, token=oauth_token.token)
+        except:
+            card = ModelCard("")
+        if card.data.tags is None:
+            card.data.tags = []
+        card.data.tags.append("llama-cpp")
+        card.data.tags.append("gguf-my-repo")
+        card.data.base_model = model_id
+        card.text = dedent(
+            f"""
+            # {new_repo_id}
+            This model was converted to GGUF format from [`{model_id}`](https://huggingface.co/{model_id}) using llama.cpp via the ggml.ai's [GGUF-my-repo](https://huggingface.co/spaces/ggml-org/gguf-my-repo) space.
+            Refer to the [original model card](https://huggingface.co/{model_id}) for more details on the model.
+            ## Use with llama.cpp
+            Install llama.cpp through brew (works on Mac and Linux)
+            ```bash
+            brew install llama.cpp
+            ```
+            Invoke the llama.cpp server or the CLI.
+            ### CLI:
+            ```bash
+            llama-cli --hf-repo {new_repo_id} --hf-file {quantized_gguf_name} -p "The meaning to life and the universe is"
+            ```
+            ### Server:
+            ```bash
+            llama-server --hf-repo {new_repo_id} --hf-file {quantized_gguf_name} -c 2048
+            ```
+            Note: You can also use this checkpoint directly through the [usage steps](https://github.com/ggerganov/llama.cpp?tab=readme-ov-file#usage) listed in the Llama.cpp repo as well.
+            Step 1: Clone llama.cpp from GitHub.
+            ```
+            git clone https://github.com/ggerganov/llama.cpp
+            ```
+            Step 2: Move into the llama.cpp folder and build it with `LLAMA_CURL=1` flag along with other hardware-specific flags (for ex: LLAMA_CUDA=1 for Nvidia GPUs on Linux).
+            ```
+            cd llama.cpp && LLAMA_CURL=1 make
+            ```
+            Step 3: Run inference through the main binary.
+            ```
+            ./llama-cli --hf-repo {new_repo_id} --hf-file {quantized_gguf_name} -p "The meaning to life and the universe is"
+            ```
+            or
+            ```
+            ./llama-server --hf-repo {new_repo_id} --hf-file {quantized_gguf_name} -c 2048
+            ```
+            """
+        )
+        card.save(f"README.md")
         if split_model:
             split_upload_model(quantized_gguf_path, new_repo_id, oauth_token, split_max_tensors, split_max_size)
         else:
             try:
+                print(f"Uploading quantized model: {quantized_gguf_path}")
                 api.upload_file(
                     path_or_fileobj=quantized_gguf_path,
                     path_in_repo=quantized_gguf_name,
         if use_imatrix and os.path.isfile(imatrix_path):
             try:
+                print(f"Uploading imatrix.dat: {imatrix_path}")
                 api.upload_file(
                     path_or_fileobj=imatrix_path,
                     path_in_repo="imatrix.dat",
             except Exception as e:
                 raise Exception(f"Error uploading imatrix.dat: {e}")
+        api.upload_file(
+            path_or_fileobj=f"README.md",
+            path_in_repo=f"README.md",
+            repo_id=new_repo_id,
+        )
+        print(f"Uploaded successfully with {imatrix_q_method if use_imatrix else q_method} option!")
         return (
             f'Find your repo <a href=\'{new_repo_url}\' target="_blank" style="text-decoration:underline">here</a>',
             "llama.png",
         return (f"Error: {e}", "error.png")
     finally:
         shutil.rmtree(model_name, ignore_errors=True)
+        print("Folder cleaned up successfully!")
 css="""/* Custom CSS to allow scrolling */
 .gradio-container {overflow-y: auto;}
 """
+with gr.Blocks(css=css) as demo:
     gr.Markdown("You must be logged in to use GGUF-my-repo.")
     gr.LoginButton(min_width=250)
         ["IQ3_M", "IQ3_XXS", "Q4_K_M", "Q4_K_S", "IQ4_NL", "IQ4_XS", "Q5_K_M", "Q5_K_S"],
         label="Imatrix Quantization Method",
         info="GGML imatrix quants type",
+        value="IQ4_NL",
         filterable=False,
         visible=False
     )
     def update_visibility(use_imatrix):
         return gr.update(visible=not use_imatrix), gr.update(visible=use_imatrix), gr.update(visible=use_imatrix)
     use_imatrix.change(
         fn=update_visibility,
         inputs=use_imatrix,
     )
 def restart_space():
+    HfApi().restart_space(repo_id="YOUR_SPACE_ID", token=HF_TOKEN, factory_reboot=True)
 scheduler = BackgroundScheduler()
 scheduler.add_job(restart_space, "interval", seconds=21600)