from __future__ import annotations import os import time import urllib.request import gradio as gr from pyharp import * from gradio_client import Client, handle_file _BACKEND_SPACE = "facebook/MelodyFlow" _BACKEND_API_NAME = "/predict" _BACKEND_TOKEN_ENV = "HF_TOKEN" _ACCEPT_USER_TOKEN = True # How many times to wake+retry a sleeping backend, and how long to wait for # it to boot (a free Space cold start can take a few minutes). _CALL_RETRIES = int(os.environ.get("BACKEND_CALL_RETRIES", "4")) _WAKE_TIMEOUT = float(os.environ.get("BACKEND_WAKE_TIMEOUT", "420")) _client = None def _backend_client(): # Lazily create and cache one warm connection using this Space's own # token (from the HF_TOKEN secret) or anonymous if none is set. User # tokens are NOT cached here -- they get a fresh per-call connection. global _client if _client is None: _token = os.environ.get(_BACKEND_TOKEN_ENV) or None _client = Client(_BACKEND_SPACE, hf_token=_token) return _client def _reset_client(): # Drop the cached connection so the next attempt reconnects to a Space # that has since finished waking. global _client _client = None def _make_conn(tok): tok = (tok or '').strip() if tok: return Client(_BACKEND_SPACE, hf_token=tok) return _backend_client() def _space_url(space): slug = space.strip().lower().replace('/', '-').replace('_', '-') return f'https://{slug}.hf.space/' def _is_cold_start(message): # Errors that mean 'the backend was asleep/booting', worth waking+retrying # (vs. a real application error, which we surface immediately). _low = (message or '').lower() return any(s in _low for s in ( 'read operation timed out', 'timed out', 'timeout', 'starting', 'building', 'not ready', 'no application', 'connection', '503', '502', )) def _wake_backend(): # A sleeping Space boots when its URL is hit; poll until it answers (or # the budget expires) so the retried call lands on a running backend. _url = _space_url(_BACKEND_SPACE) _deadline = time.time() + _WAKE_TIMEOUT _delay = 5.0 while time.time() < _deadline: try: _req = urllib.request.Request(_url, headers={'User-Agent': 'harp-frontend'}) with urllib.request.urlopen(_req, timeout=30) as _resp: if getattr(_resp, 'status', 200) < 500: return True except Exception: pass time.sleep(_delay) _delay = min(_delay * 1.5, 30.0) return False def _quota_hint(message): # Turn a backend ZeroGPU quota error into an actionable message. # NOTE: 'message' is the backend's error text; it never contains our token. _low = (message or "").lower() if "quota" in _low or "zerogpu" in _low: if _ACCEPT_USER_TOKEN: return ( "The backend's ZeroGPU quota is exhausted for the identity making " "this call. Paste your own Hugging Face token in the token field " "(read scope) so usage is attributed to your account." ) return ( "The backend's ZeroGPU quota is exhausted. This Space's calls are " "anonymous unless an HF_TOKEN secret is set (Settings -> Variables " "and secrets); use a token from a PRO account or a ZeroGPU-enabled org." ) return message or "Backend call failed." model_card = ModelCard( name="Melodyflow", description="TODO: describe this model.", author="facebook", tags=[], ) def process_fn(text, steps, target_flowstep, regularize, regularization_strength, duration, melody, _hf_user_token=''): _tok = (_hf_user_token or '').strip() # Call the backend, waking it and retrying if it was asleep (a cold # start otherwise fails the first hit with 'read operation timed out'). _raw = None for _attempt in range(_CALL_RETRIES + 1): try: _conn = _make_conn(_tok) _raw = _conn.predict( 'facebook/melodyflow-t24-30secs', text, 'midpoint', steps, target_flowstep, regularize, regularization_strength, duration, (handle_file(melody) if melody else None), api_name="/predict", ) break except Exception as _exc: # never surfaces the token if _attempt < _CALL_RETRIES and _is_cold_start(str(_exc)): _reset_client() _wake_backend() continue raise gr.Error(_quota_hint(str(_exc))) _values = list(_raw) if isinstance(_raw, (list, tuple)) else [_raw] _detail = " | ".join(str(_v) for _v in _values if isinstance(_v, str) and _v.strip()) _out_generated_audio_variation_1 = _values[0] if len(_values) > 0 else None if not _out_generated_audio_variation_1: raise gr.Error(_detail or "The backend Space returned no 'generated_audio_variation_1' output. Check the backend Space's logs; if it uses ZeroGPU it may need a moment to warm up.") _out_generated_audio_variation_2 = _values[1] if len(_values) > 1 else None if not _out_generated_audio_variation_2: raise gr.Error(_detail or "The backend Space returned no 'generated_audio_variation_2' output. Check the backend Space's logs; if it uses ZeroGPU it may need a moment to warm up.") _out_generated_audio_variation_3 = _values[2] if len(_values) > 2 else None if not _out_generated_audio_variation_3: raise gr.Error(_detail or "The backend Space returned no 'generated_audio_variation_3' output. Check the backend Space's logs; if it uses ZeroGPU it may need a moment to warm up.") return _out_generated_audio_variation_1, _out_generated_audio_variation_2, _out_generated_audio_variation_3 with gr.Blocks() as demo: input_components = [ gr.Textbox(label="Input Text"), gr.Slider(minimum=0.0, maximum=1.0, step=0.1, value=128.0, label="Inference steps"), gr.Slider(minimum=0.0, maximum=1.0, step=0.1, value=0.0, label="Target Flow step"), gr.Checkbox(value=False, label="Regularize"), gr.Slider(minimum=0.0, maximum=1.0, step=0.1, value=0.2, label="Regularization Strength"), gr.Slider(minimum=0.0, maximum=1.0, step=0.1, value=30.0, label="Duration"), gr.Audio(type="filepath", label="File or Microphone"), gr.Textbox(label="Hugging Face token (optional)", type="password", info="Optional. Paste a Hugging Face token (Settings -> Access Tokens, read scope) so ZeroGPU usage on the backend is charged to YOUR account. Used only for this call; not stored. Leave blank to use this Space's own token."), ] output_components = [ gr.Audio(type="filepath", label="Generated Audio - variation 1"), gr.Audio(type="filepath", label="Generated Audio - variation 2"), gr.Audio(type="filepath", label="Generated Audio - variation 3"), ] build_endpoint( model_card=model_card, input_components=input_components, output_components=output_components, process_fn=process_fn, ) demo.queue().launch(share=True, show_error=False, pwa=True)