[英]Preprocessing data: to remove italian stopwords for text analysis
! pip install stop-words
from stop_words import get_stop_words
stop = get_stop_words('italian')
import re
# helper function to clean tweets
def processTweet(tweet):
# Remove HTML special entities (e.g. &)
tweet = re.sub(r'\&\w*;', '', tweet)
#Convert @username to AT_USER
tweet = re.sub('@[^\s]+','',tweet)
# Remove tickers
tweet = re.sub(r'\$\w*', '', tweet)
# To lowercase
tweet = tweet.lower()
# Remove hyperlinks
tweet = re.sub(r'https?:\/\/.*\/\w*', '', tweet)
# Remove hashtags
tweet = re.sub(r'#\w*', '', tweet)
# Remove Punctuation and split 's, 't, 've with a space for filter
tweet = ' '.join(re.sub("(@[A-Za-z0-9]+)|(#)|(\w+:\/\/\S+)|(\S*\d\S*)|([,;.?!:])",
" ", tweet).split())
#tweet = re.sub(r'[' + punctuation.replace('@', '') + ']+', ' ', tweet)
# Remove words with 2 or fewer letters
tweet = re.sub(r'\b\w{1,3}\b', '', tweet)
# Remove whitespace (including new line characters)
tweet = re.sub(r'\s\s+', ' ', tweet)
# Remove single space remaining at the front of the tweet.
tweet = tweet.lstrip(' ')
# Remove characters beyond Basic Multilingual Plane (BMP) of Unicode:
tweet = ''.join(c for c in tweet if c <= '\uFFFF')
return tweet
df['text'] = df['text'].apply(processTweet)
只需使用你一直在使用的 re.sub() :
exclusions = '|'.join(stop)
tweet = re.sub(exclusions, '', tweet)
考慮以下示例
import re
stops = ["and","or","not"] # list of words to remove
text = "Band and nothing else!" # and in Band and not in nothing should stay
pattern = r'\b(?:' + '|'.join(re.escape(s) for s in stops) + r')\b'
clean = re.sub(pattern, '', text)
print(clean)
輸出
Band nothing else!
說明: re.escape
處理在正則表達式模式中具有特殊含義的字符(例如.
)並將它們轉換為文字版本(因此re.escape(".")
匹配文字.
不是任何字符), |
是替代方法,使用所有單詞的連接替代方法是構建, (?:
... )
是非捕獲組,它允許我們在開始時使用一個\b
\b
在結尾處使用一個 \b 而不是每個單詞。 \b
是單詞邊界,這里用於確保僅刪除整個單詞,而不是例如Band
變成B
。
聲明:本站的技術帖子網頁,遵循CC BY-SA 4.0協議,如果您需要轉載,請注明本站網址或者原文地址。任何問題請咨詢:yoyou2525@163.com.