fx old examples

2026-02-05 18:09:22 +08:00 · 2024-09-24 12:01:47 +00:00
parent 1d48167271
commit fa8036ae1c
1 changed files with 23 additions and 27 deletions
--- a/examples/colab_record_example.ipynb
+++ b/examples/colab_record_example.ipynb
@@ -17,6 +17,7 @@
   },
   "outputs": [],
   "source": [
+    "#!apt install ffmpeg\n",
    "!pip -q install pydub\n",
    "from google.colab import output\n",
    "from base64 import b64decode, b64encode\n",
@@ -37,13 +38,12 @@
    "                              model='silero_vad',\n",
    "                              force_reload=True)\n",
    "\n",
-    "def int2float(sound):\n",
-    "    abs_max = np.abs(sound).max()\n",
-    "    sound = sound.astype('float32')\n",
-    "    if abs_max > 0:\n",
-    "        sound *= 1/32768\n",
-    "    sound = sound.squeeze()\n",
-    "    return sound\n",
+    "def int2float(audio):\n",
+    "    samples = audio.get_array_of_samples()\n",
+    "    new_sound = audio._spawn(samples)\n",
+    "    arr = np.array(samples).astype(np.float32)\n",
+    "    arr = arr / np.abs(arr).max()\n",
+    "    return arr\n",
    "\n",
    "AUDIO_HTML = \"\"\"\n",
    "<script>\n",
@@ -133,7 +133,7 @@
    "    audio.export('test.mp3', format='mp3')\n",
    "    audio = audio.set_channels(1)\n",
    "    audio = audio.set_frame_rate(16000)\n",
-    "    audio_float = int2float(np.array(audio.get_array_of_samples()))\n",
+    "    audio_float = int2float(audio)\n",
    "    audio_tens = torch.tensor(audio_float)\n",
    "    return audio_tens\n",
    "\n",
@@ -158,8 +158,7 @@
    "        line.set_data(x, y)\n",
    "        line.set_color('#990000')\n",
    "        return line,\n",
-    "\n",
-    "    anim = FuncAnimation(fig, animate, init_func=init, interval=interval, save_count=audio_duration / (interval / 1000))\n",
+    "    anim = FuncAnimation(fig, animate, init_func=init, interval=interval, save_count=int(audio_duration / (interval / 1000)))\n",
    "\n",
    "    f = r\"animation.mp4\"\n",
    "    writervideo = FFMpegWriter(fps=1000/interval)\n",
@@ -174,15 +173,10 @@
    "\n",
    "def record_make_animation():\n",
    "  tensor = record()\n",
-    "\n",
    "  print('Calculating probabilities...')\n",
    "  speech_probs = []\n",
    "  window_size_samples = 512\n",
-    "  for i in range(0, len(tensor), window_size_samples):\n",
-    "      if len(tensor[i: i+ window_size_samples]) < window_size_samples:\n",
-    "        break\n",
-    "      speech_prob = model(tensor[i: i+ window_size_samples], 16000).item()\n",
-    "      speech_probs.append(speech_prob)\n",
+    "  speech_probs = model.audio_forward(tensor, sr=16000)[0].tolist()\n",
    "  model.reset_states()\n",
    "  print('Making animation...')\n",
    "  make_animation(speech_probs, len(tensor) / 16000)\n",
@@ -196,7 +190,9 @@
    "  <video width=800 controls>\n",
    "        <source src=\"%s\" type=\"video/mp4\">\n",
    "  </video>\n",
-    "  \"\"\" % data_url))"
+    "  \"\"\" % data_url))\n",
+    "\n",
+    "  return speech_probs"
   ]
  },
  {
@@ -216,7 +212,7 @@
   },
   "outputs": [],
   "source": [
-    "record_make_animation()"
+    "speech_probs = record_make_animation()"
   ]
  }
 ],