Spaces:

Akjava
/

chat-phi-4-deepseek-R1K-RL-EZO

Runtime error

App Files Files Community

Akjava commited on Jan 31, 2025

Commit

c411706

verified ·

1 Parent(s): 0e10553

Update app.py

Browse files

Files changed (1) hide show

app.py +8 -38

app.py CHANGED Viewed

@@ -8,16 +8,14 @@ from threading import Thread
 import gradio as gr
 text_generator = None
-is_hugging_face = True
 model_id = "AXCXEPT/phi-4-deepseek-R1K-RL-EZO"
 #model_id = "AXCXEPT/phi-4-open-R1-Distill-EZOv1"
 huggingface_token = os.getenv("HUGGINGFACE_TOKEN")
-#huggingface_token = None
 device = "auto" # torch.device("cuda" if torch.cuda.is_available() else "cpu")
 device = "cuda"
 dtype = torch.bfloat16
-#dtype = torch.float16
 if not huggingface_token:
     pass
@@ -32,36 +30,14 @@ if not huggingface_token:
 tokenizer = AutoTokenizer.from_pretrained(model_id, token=huggingface_token)
-print(tokenizer.special_tokens_map)
 # 特殊トークンIDを確認
-print(tokenizer.eos_token_id)
-print(tokenizer.encode("<|im_end|>", add_special_tokens=False))
-print(model_id,device,dtype)
 histories = []
-#model = None
-if not is_hugging_face:
-    model = AutoModelForCausalLM.from_pretrained(
-        model_id, token=huggingface_token ,torch_dtype=dtype,device_map=device
-    )
-    text_generator = pipeline("text-generation", model=model, tokenizer=tokenizer,torch_dtype=dtype,device_map=device,stream=True ) #pipeline has not to(device)
-    if next(model.parameters()).is_cuda:
-        print("The model is on a GPU")
-    else:
-        print("The model is on a CPU")
-    #print(f"text_generator.device='{text_generator.device}")
-    if str(text_generator.device).strip() == 'cuda':
-        print("The pipeline is using a GPU")
-    else:
-        print("The pipeline is using a CPU")
-print("initialized")
 def generate_text(messages):
@@ -81,18 +57,12 @@ def generate_text(messages):
     generated_output = ""
     thread.start()
     for new_text in streamer:
-        generated_output += new_text.replace("<|im_end|>","")
         yield generated_output
-#generate_text.zerogpu = True
 @spaces.GPU(duration=120)
-def call_generate_text(message, history):
-   # history.append({"role": "user", "content": message})
-    #print(message)
-    #print(history)
     messages = history+[{"role":"user","content":message}]
     try:

 import gradio as gr
 text_generator = None
 model_id = "AXCXEPT/phi-4-deepseek-R1K-RL-EZO"
 #model_id = "AXCXEPT/phi-4-open-R1-Distill-EZOv1"
 huggingface_token = os.getenv("HUGGINGFACE_TOKEN")
 device = "auto" # torch.device("cuda" if torch.cuda.is_available() else "cpu")
 device = "cuda"
 dtype = torch.bfloat16
 if not huggingface_token:
     pass
 tokenizer = AutoTokenizer.from_pretrained(model_id, token=huggingface_token)
+#print(tokenizer.special_tokens_map)
 # 特殊トークンIDを確認
+#print(tokenizer.eos_token_id)
+#print(tokenizer.encode("<|im_end|>", add_special_tokens=False))
+#print(model_id,device,dtype)
 histories = []
 def generate_text(messages):
     generated_output = ""
     thread.start()
     for new_text in streamer:
+        generated_output += new_text.replace("<|im_end|>","")#just replace
         yield generated_output
+# SDK version is very important in README.md
 @spaces.GPU(duration=120)
+def call_generate_text(message, history):
     messages = history+[{"role":"user","content":message}]
     try: