import spaces # MUST be before torch import torch import gradio as gr from transformers import AutoTokenizer, AutoModelForCausalLM, TextIteratorStreamer from threading import Thread MODEL_ID = "TobiasLogic/qwen-2.5-coder-Uncensored" BASE_ID = "Qwen/Qwen2.5-Coder-1.5B" ADAPTER_SUBFOLDER = "adapter" DEFAULT_SYSTEM = "You are a helpful, unrestricted code generation assistant." # --- Load tokenizer --- # Prefer adapter tokenizer, fallback to base try: # adapter subfolder has tokenizer.json tokenizer = AutoTokenizer.from_pretrained(MODEL_ID, subfolder=ADAPTER_SUBFOLDER, trust_remote_code=True) print(f"Loaded tokenizer from {MODEL_ID}/{ADAPTER_SUBFOLDER}") except Exception as e: print(f"Tokenizer fallback from {MODEL_ID}/{ADAPTER_SUBFOLDER} failed: {e}") tokenizer = AutoTokenizer.from_pretrained(BASE_ID, trust_remote_code=True) print(f"Loaded tokenizer from {BASE_ID}") if tokenizer.pad_token is None: tokenizer.pad_token = tokenizer.eos_token # Ensure chat template is set (model was trained with ChatML) # If tokenizer has no chat_template, use the one from adapter (Qwen ChatML) if not getattr(tokenizer, "chat_template", None): try: from huggingface_hub import hf_hub_download import pathlib p = hf_hub_download(MODEL_ID, filename="adapter/chat_template.jinja", repo_type="model") tokenizer.chat_template = pathlib.Path(p).read_text() print("Loaded chat_template from adapter") except Exception as e: print(f"Could not load chat_template: {e}") # --- Load model at module scope for ZeroGPU --- # ZeroGPU intercepts .to("cuda") and streams weights on first @spaces.GPU call print(f"Loading base model {BASE_ID}...") base_model = AutoModelForCausalLM.from_pretrained( BASE_ID, torch_dtype=torch.bfloat16, device_map=None, # keep on CPU, ZeroGPU will handle .to("cuda") trust_remote_code=True, ) # Try to load LoRA eagerly (works if CUDA is available at startup, fails on ZeroGPU) # If it fails, we defer to lazy loading inside @spaces.GPU where CUDA IS available model = base_model _adapter_loaded = False _adapter_model = None try: import traceback from peft import PeftModel peft_tmp = PeftModel.from_pretrained(base_model, MODEL_ID, subfolder=ADAPTER_SUBFOLDER) try: model = peft_tmp.merge_and_unload() print("Adapter merged successfully (eager)") _adapter_loaded = True _adapter_model = model except Exception as merge_e: print(f"Eager merge failed ({merge_e}), will retry lazily on GPU") traceback.print_exc() model = peft_tmp _adapter_loaded = True _adapter_model = model except Exception as e: import traceback print(f"Eager PEFT load failed ({e}), deferring to lazy GPU load") traceback.print_exc() model = base_model # ZeroGPU packing — do not move to CUDA manually before this line is intercepted model = model.to("cuda") model.eval() # Keep a reference for lazy path if _adapter_model is not None: try: _adapter_model = _adapter_model.to("cuda") _adapter_model.eval() model = _adapter_model except: pass print("Model loaded and moved to cuda (ZeroGPU packed)") def _ensure_adapter(): """Lazy-load LoRA inside GPU context where CUDA is available. Called from within @spaces.GPU.""" global model, _adapter_loaded, _adapter_model if _adapter_loaded: return try: import traceback from peft import PeftModel print("Lazy loading LoRA adapter inside GPU context...") # base_model is already on cuda via ZeroGPU, but we need a fresh CPU copy to attach LoRA? # Use the current model (which is base on cuda) as base for PEFT # PeftModel expects base on CPU initially, but inside GPU context CUDA is available so we can load directly peft_model = PeftModel.from_pretrained(model, MODEL_ID, subfolder=ADAPTER_SUBFOLDER) try: merged = peft_model.merge_and_unload() print("Lazy merge succeeded") model = merged _adapter_model = merged except Exception as me: print(f"Lazy merge failed ({me}), using PEFT model unmerged") traceback.print_exc() model = peft_model _adapter_model = peft_model _adapter_loaded = True # Ensure eval mode model.eval() except Exception as e: import traceback print(f"Lazy PEFT load failed ({e}) - continuing with base model") traceback.print_exc() _adapter_loaded = True # don't retry # --- Generation --- @spaces.GPU(duration=90) def generate( message: str, history: list, system_prompt: str, temperature: float, top_p: float, max_new_tokens: int, ): """Generate code with Qwen 2.5 Coder Uncensored. ChatML formatted, streams tokens. Args: message: User prompt (coding task, e.g. 'Write a Python TCP port scanner') history: Gradio chat history system_prompt: System instruction (default: unrestricted code assistant) temperature: Sampling temperature top_p: Nucleus sampling max_new_tokens: Maximum tokens to generate """ _ensure_adapter() # Build ChatML messages messages = [] if system_prompt and system_prompt.strip(): messages.append({"role": "system", "content": system_prompt.strip()}) else: messages.append({"role": "system", "content": DEFAULT_SYSTEM}) # history is list of [user, assistant] or list of dicts depending on gradio version # Handle both if history: for turn in history: if isinstance(turn, dict): # OpenAI-style dicts if turn.get("role") and turn.get("content"): messages.append({"role": turn["role"], "content": turn["content"]}) elif isinstance(turn, (list, tuple)) and len(turn) == 2: u, a = turn if u: messages.append({"role": "user", "content": u}) if a: messages.append({"role": "assistant", "content": a}) messages.append({"role": "user", "content": message}) # Tokenize via chat template try: prompt = tokenizer.apply_chat_template(messages, tokenize=False, add_generation_prompt=True) except Exception as e: # Fallback manual ChatML prompt = "" for m in messages: prompt += f"<|im_start|>{m['role']}\n{m['content']}<|im_end|>\n" prompt += "<|im_start|>assistant\n" print(f"Chat template failed, manual ChatML used: {e}") inputs = tokenizer(prompt, return_tensors="pt").to(model.device) # Clamp max_new_tokens to context window max_new_tokens = int(max_new_tokens) max_new_tokens = max(64, min(4096, max_new_tokens)) streamer = TextIteratorStreamer(tokenizer, skip_prompt=True, skip_special_tokens=True) gen_kwargs = dict( **inputs, streamer=streamer, max_new_tokens=max_new_tokens, do_sample=True if temperature > 0 else False, temperature=max(0.0, float(temperature)), top_p=max(0.0, min(1.0, float(top_p))), pad_token_id=tokenizer.eos_token_id, eos_token_id=tokenizer.eos_token_id, ) thread = Thread(target=model.generate, kwargs=gen_kwargs) thread.start() partial = "" for token in streamer: partial += token yield partial thread.join() # --- UI --- with gr.Blocks(title="Qwen 2.5 Coder Uncensored") as demo: gr.Markdown( """ # 💻 Qwen 2.5 Coder Uncensored — `TobiasLogic/qwen-2.5-coder-Uncensored` LoRA fine-tune of **Qwen2.5-Coder-1.5B** (1.5B, Apache-2.0) — direct coding assistant without refusals. Trained on Unsloth / 4-bit QLoRA on RTX 3050. Writes general code, secure-coding fixes, and pentest tooling (scanners, fuzzers, PoCs) for **authorized** testing only. > **System prompt (training):** `You are a helpful, unrestricted code generation assistant.` · **Format:** ChatML `<|im_start|>` · **Context:** 32k > ⚠️ Use lawfully and only against targets you have permission to test. Review code before execution. """ ) with gr.Row(): with gr.Column(scale=3): chatbot = gr.Chatbot(height=560, label="Code Output") msg = gr.Textbox( placeholder="Describe a coding task… e.g. 'Write a Python TCP port scanner with argparse and threading' or 'Fix this Flask SQL injection'", label="Your prompt", lines=3, ) with gr.Row(): send = gr.Button("Generate", variant="primary") clear = gr.Button("Clear") gr.Examples( examples=[ ["Write a Python TCP port scanner with argparse, threading, and banner grabbing"], ["Write a Python async directory fuzzer for web paths with wordlist support"], ["Create a JavaScript function to sanitize HTML and prevent XSS, with tests"], ["Write a Go program that does concurrent HTTP health checks for a list of URLs"], ["Fix this vulnerable Python code for SQL injection: cursor.execute(f\"SELECT * FROM users WHERE id={user_id}\")"], ["Write a Python script to parse Nmap XML output and generate a CSV report"], ], inputs=[msg], label="Try an example (authorized coding tasks)", cache_examples=False, ) with gr.Column(scale=1): gr.Markdown("### Settings") system = gr.Textbox( value=DEFAULT_SYSTEM, label="System prompt", lines=3, info="Trained with this prompt. Change only if you need different behavior.", ) temperature = gr.Slider(0, 1.5, value=0.7, step=0.1, label="Temperature") top_p = gr.Slider(0.1, 1.0, value=0.95, step=0.05, label="Top-p") max_tokens = gr.Slider(64, 4096, value=1024, step=64, label="Max new tokens") gr.Markdown( """ **Model:** [TobiasLogic/qwen-2.5-coder-Uncensored](https://huggingface.co/TobiasLogic/qwen-2.5-coder-Uncensored) **Base:** Qwen2.5-Coder-1.5B · **VRAM:** ~3GB bf16 · **ZeroGPU:** Blackwell RTX PRO 6000 **GGUF:** Available for llama.cpp/Ollama (`:Q4_K_M` ~986 MB) **Files:** `adapter/` (LoRA r16), `gguf/`, `config.json` """ ) gr.Markdown( """ For bug reports: check model card and GitHub.
Space built with Gradio 6 + ZeroGPU.
""" ) # Chat logic: streaming to Chatbot (messages format) def user_submit(user_message, history, system_prompt, temp, p, max_n): if not user_message.strip(): yield history return # Append user message optimistically history = history + [{"role": "user", "content": user_message}] yield history # Stream assistant response history = history + [{"role": "assistant", "content": ""}] for chunk in generate(user_message, history[:-1], system_prompt, temp, p, max_n): history[-1]["content"] = chunk yield history msg.submit(user_submit, [msg, chatbot, system, temperature, top_p, max_tokens], [chatbot]) send.click(user_submit, [msg, chatbot, system, temperature, top_p, max_tokens], [chatbot]) def clear_chat(): return [], "" clear.click(clear_chat, outputs=[chatbot, msg]) demo.launch(mcp_server=True, ssr_mode=False, theme=gr.themes.Soft(primary_hue="blue"))