• דף הבית
    • אינדקס קישורים
    • פוסטים אחרונים
    • משתמשים
    • חיפוש בהגדרות המתקדמות
    • חיפוש גוגל בפורום
    • ניהול המערכת
    • ניהול המערכת - שרת private
    • הרשמה
    • התחברות

    תגובה: 🎙️🎯 זיהוי דיבור בעברית – חינם, איכותי, מדויק!!

    מתוזמן נעוץ נעול הועבר עזרה הדדית למשתמשים מתקדמים
    2 פוסטים 2 כותבים 39 צפיות 2 עוקבים
    טוען פוסטים נוספים
    • מהישן לחדש
    • מהחדש לישן
    • הכי הרבה הצבעות
    תגובה
    • תגובה כנושא
    התחברו כדי לפרסם תגובה
    נושא זה נמחק. רק משתמשים עם הרשאות מתאימות יוכלו לצפות בו.
    • י מנותק
      יעקב יצחק
      נערך לאחרונה על ידי יעקב יצחק

      תגובה: 🎙️🎯 זיהוי דיבור בעברית – חינם, איכותי, מדויק!!
      ניסיתי לכתוב כזה קוד - אבל דרך הוויספר של גרוק
      קוד גרוק.py
      מה דעתכם?
      שימו לב שצריך מפתח API של GROK בשורה 21

      import os
      import tempfile
      import logging
      import requests
      from flask import Flask, request, jsonify
      from pydub import AudioSegment
      from rapidfuzz import process, fuzz
      from groq import Groq
      
      # ------------------ Logging Configuration ------------------
      logging.basicConfig(
          level=logging.INFO,
          format="%(asctime)s | %(levelname)s | %(message)s",
          datefmt="%H:%M:%S"
      )
      
      app = Flask(__name__)
      
      # Initialize Groq client (Make sure GROQ_API_KEY environment variable is set)
      # export GROQ_API_KEY="your_api_key_here"
      client = Groq(api_key=os.environ.get("GROQ_API_KEY"))
      
      # List of possible keywords to match
      KEYWORDS = ["בני ברק", "ירושלים", "תל אביב", "חיפה", "אשדוד"]
      
      # ------------------ Helper Functions ------------------
      
      def add_silence(input_path: str) -> AudioSegment:
          """
          Add one second of silence at the beginning and end of the audio file.
          This improves speech recognition accuracy, especially for short recordings.
          """
          logging.info("Adding one second of silence to audio file...")
          audio = AudioSegment.from_file(input_path)
          silence = AudioSegment.silent(duration=1000)  # 1000ms = 1 second
          return silence + audio + silence
      
      def recognize_speech_groq(audio_path: str) -> str:
          """
          Perform speech recognition using Groq's Whisper-v3 model.
          """
          try:
              logging.info("Sending audio to Groq API (whisper-large-v3)...")
              with open(audio_path, "rb") as file:
                  transcription = client.audio.transcriptions.create(
                    file=(os.path.basename(audio_path), file.read()),
                    model="whisper-large-v3",
                    language="he", # Specify Hebrew to guide the model
                    response_format="json"
                  )
              
              text = transcription.text
              logging.info(f"Recognized text: {text}")
              return text
          except Exception as e:
              logging.error(f"Error during speech recognition with Groq: {e}")
              return ""
      
      def find_best_match(text: str) -> str | None:
          """
          Find the closest matching word from the predefined KEYWORDS list.
          """
          if not text:
              return None
      
          result = process.extractOne(text, KEYWORDS, scorer=fuzz.ratio)
          if result and result[1] >= 80:
              logging.info(f"Best match found: {result[0]} (confidence: {result[1]}%)")
              return result[0]
      
          logging.info("No sufficient match found.")
          return None
      
      # ------------------ API Endpoint ------------------
      
      @app.route("/upload_audio", methods=["GET"])
      def upload_audio():
          """
          Endpoint to receive an audio file via GET parameter,
          download it, process it, and return the recognized text with the best match.
          Example usage:
          /upload_audio?file_url=https://example.com/audio.wav
          """
          file_url = request.args.get("file_url")
          if not file_url:
              logging.error("Missing 'file_url' parameter.")
              return jsonify({"error": "Missing 'file_url' parameter"}), 400
      
          logging.info(f"Received file URL: {file_url}")
      
          try:
              # Step 1: Download the audio file
              response = requests.get(file_url, timeout=15)
              if response.status_code != 200:
                  logging.error(f"Failed to download audio file. Status code: {response.status_code}")
                  return jsonify({"error": "Failed to download audio file"}), 400
      
              # Create a temporary directory to handle the files
              with tempfile.TemporaryDirectory() as temp_dir:
                  temp_input_path = os.path.join(temp_dir, "input_audio")
                  temp_processed_path = os.path.join(temp_dir, "processed_audio.wav")
                  
                  # Save downloaded file
                  with open(temp_input_path, "wb") as f:
                      f.write(response.content)
                  logging.info(f"Audio downloaded and saved temporarily.")
      
                  # Step 2: Add silence and export as WAV (Whisper accepts specific formats like wav, mp3, m4a)
                  processed_audio = add_silence(temp_input_path)
                  processed_audio.export(temp_processed_path, format="wav")
      
                  # Step 3: Speech recognition using Groq
                  recognized_text = recognize_speech_groq(temp_processed_path)
      
                  # Step 4: Matching against predefined keywords
                  matched_word = find_best_match(recognized_text)
      
                  if matched_word:
                      logging.info(f"Final matched keyword: {matched_word}")
                  else:
                      logging.info("No keyword match found.")
      
          except Exception as e:
              logging.error(f"Processing error: {e}")
              return jsonify({"error": "Error processing the audio file"}), 500
      
          return jsonify({
              "recognized_text": recognized_text,
              "matched_word": matched_word if matched_word else "No match found"
          })
      
      # ------------------ Run Server ------------------
      
      if __name__ == "__main__":
          port = int(os.environ.get("PORT", 5000))
          logging.info(f"Server running on port {port}")
          app.run(host="0.0.0.0", port=port)
      
      ט תגובה 1 תגובה אחרונה תגובה ציטוט 1
      • ט מנותק
        טנטפון @יעקב יצחק
        נערך לאחרונה על ידי

        @יעקב-יצחק מצוין אבל API של גרוק דורש תשלום על הסימונים זכור לי תתקן אותי אם לא

        תגובה 1 תגובה אחרונה תגובה ציטוט 0

        שלום! נראה שהשיחה הזו מעניינת אותך, אבל עדיין אין לך חשבון.

        נמאס לכם לגלול בין אותם הפוסטים בכל ביקור? כשנרשמים לחשבון, תמיד תחזרו בדיוק למקום שבו הייתם קודם, ותוכלו לבחור לקבל התראות על תגובות חדשות (בין אם במייל, ובין אם בהתראת פוש). תוכלו גם לשמור סימניות ולפרגן ב-upvote לפוסטים כדי להביע הערכה לחברי קהילה אחרים.

        בעזרת התרומה שלך, הפוסט הזה יכול להיות אפילו טוב יותר 💗

        הרשמה התחברות
        • פוסט ראשון
          פוסט אחרון