I am using a transformer linked to an ML model to check if a particular video contains any offensive language. First, I have converted the video file to audio and then using Watson library it is converted to text. However, it is giving me errors when I try to check if the text is offensive or not.
import subprocess
from ibm_watson import SpeechToTextV1
from ibm_watson.websocket import RecognizeCallback, AudioSource
from ibm_cloud_sdk_core.authenticators import IAMAuthenticator
import speech_recognition as sr
import os
from happytransformer import HappyTextClassification
command='ffmpeg -i abuses.mkv -ab 160k -ar 44100 -vn abusesaudio.wav'
subprocess.call(command, shell=True)
apikey= 'LSwQMGtTGAuA1PPeJVWmcyJMpG3cNI7sIqiAOJjytGMV'
url= 'https://api.us-south.speech-to-text.watson.cloud.ibm.com/instances/92df7d36-ec58-478a-ac0e-c496068ca863'
authenticator= IAMAuthenticator(apikey)
stt= SpeechToTextV1(authenticator=authenticator)
stt.set_service_url(url)
with open ('abusesaudio.wav','rb') as f:
res= stt.recognize(audio=f, content_type='audio/wav', model='en-AU_NarrowbandModel', continuous=True).get_result()
res2=str(res)
print(res2)
print(type(res2))
text=[result['alternatives'][0]['transcript'].rstrip() + '.\n' for result in res['results']]
text=[para[0].title()+para[1:] for para in text]
transcript= ''.join(text)
with open ('aubuses text.txt','w') as out:
out.writelines(transcript)
happy_tc= HappyTextClassification("BERT", "cardiffnlp/twitter-roberta-base-offensive", 2)
result = happy_tc.classify_text(res2)
print(result)
print(type(result.label))
print(type(result.score))
if result.label =="LABEL_1":
print("Text is offensive")
offensepercent= result.score*float(100)
print(offensepercent)
the error is in the line "result = happy_tc.classify_text(res2)"