Files
solveig_master/root_word_frequency_plot.ipynb
T
2023-05-23 22:10:06 +02:00

397 KiB
Raw Blame History

In [1]:
# load in data
import pandas as pd
from pathlib import Path

tweet_df = pd.read_csv('tweets_relevance_list.csv', index_col=0)

# remove rows with score of zero
tweet_df = tweet_df[tweet_df['score'].gt(0)]

# remove wrong folder
tweet_df['path'] = tweet_df['path'].str.replace('/home/brian/solveig_master/', '')

# ensure datatype
tweet_df['path'] = tweet_df['path'].apply(Path)

tweet_df
Out [1]:
path score
0 tweets/2021-08-24-07-40.txt 15
1 tweets/2021-06-22-13-19.txt 14
2 tweets/2020-07-13-10-03.txt 14
3 tweets/2022-04-28-12-38.txt 13
4 tweets/2022-10-20-14-08.txt 13
... ... ...
1331 tweets/2016-11-08-13-35.txt 1
1332 tweets/2022-06-16-12-15.txt 1
1333 tweets/2016-05-12-12-48.txt 1
1334 tweets/2016-06-01-11-03.txt 1
1335 tweets/2014-01-28-08-50.txt 1

1336 rows × 2 columns

In [2]:
# load in word root list
word_root_df = pd.read_csv('word_root_ranking.csv', index_col=0)

# parse string-lists
word_root_df['word_matches'] = word_root_df['word_matches'].str.strip("[]'").str.split("', '")
# word_list = sum(word_list, []) # flatten nested list

word_root_df
Out [2]:
root mean_ranking std_ranking word_matches
0 challeng 420 264 [challenges, challenge, challenging]
1 SDG 423 0 [sdgs]
2 ESG 636 0 [esg]
3 recycling 860 154 [recycling, shiprecycling]
4 reduc 1242 965 [reduce, reducing, reduced, reduction, reducti...
5 planet 1295 0 [planet]
6 CSR 1371 0 [csr]
7 sustainab 1527 1873 [sustainability, sustainable, sustainableshipp...
8 clean 1695 837 [theoceancleanup, cleanup, clean, cleanseas, o...
9 methanol 1700 1044 [methanol, emethanol]
10 future 1770 1684 [future, futureproofing]
11 garbage 1808 612 [garbage, greatpacificgarbagepatch]
12 responsib 1889 1135 [responsible, responsibility, responsibly, res...
13 carb 1952 1526 [decarbonisation, carbon, decarbonization, car...
14 chang 2148 1667 [change, changing, climatechange, changes, exc...
15 ocean 2169 1629 [ocean, theoceancleanup, oceans, oceanplastic,...
16 plastic 2320 1490 [plastic, oceanplastic, plasticwaste, plasticp...
17 neutral 2382 1798 [neutral, carbonneutral, neutrality, co2neutral]
18 environment 2391 1574 [environment, environmental, unenvironment, en...
19 green 2393 1158 [green, greenfuels, greener, greenfuel, greent...
20 sulphur 2828 1772 [sulphur, lowsulphur]
21 emissions 2925 2046 [emissions, carbonemissions, zeroemissions]
22 zero 2977 1646 [zero, netzero, zerocarbon, zerocarbonship, ze...
23 climate 3046 2095 [climateaction, climate, climatechange, climat...
24 eco 3074 1269 [ecosystem, eco, maerskecodelivery, ecofriendl...
25 mission 3247 1950 [emissions, mission, eucommission, carbonemiss...
26 CO2 3390 2275 [co2, co2neutral, co2emission]
27 bio 3687 1274 [biofuel, biofuels, biohuts, biodiversity]
In [3]:
# define time bins
period_range = pd.period_range(start='2011-01-01', end='2023-03-31', freq='M')

period_range
Out [3]:
PeriodIndex(['2011-01', '2011-02', '2011-03', '2011-04', '2011-05', '2011-06',
             '2011-07', '2011-08', '2011-09', '2011-10',
             ...
             '2022-06', '2022-07', '2022-08', '2022-09', '2022-10', '2022-11',
             '2022-12', '2023-01', '2023-02', '2023-03'],
            dtype='period[M]', length=147)
In [4]:
# prepare dataframe
df = pd.DataFrame(index=period_range, columns=word_root_df['root'])
df.iloc[:, :] = 0 # set all to 0

df
Out [4]:
root challeng SDG ESG recycling reduc planet CSR sustainab clean methanol ... environment green sulphur emissions zero climate eco mission CO2 bio
2011-01 0 0 0 0 0 0 0 0 0 0 ... 0 0 0 0 0 0 0 0 0 0
2011-02 0 0 0 0 0 0 0 0 0 0 ... 0 0 0 0 0 0 0 0 0 0
2011-03 0 0 0 0 0 0 0 0 0 0 ... 0 0 0 0 0 0 0 0 0 0
2011-04 0 0 0 0 0 0 0 0 0 0 ... 0 0 0 0 0 0 0 0 0 0
2011-05 0 0 0 0 0 0 0 0 0 0 ... 0 0 0 0 0 0 0 0 0 0
... ... ... ... ... ... ... ... ... ... ... ... ... ... ... ... ... ... ... ... ... ...
2022-11 0 0 0 0 0 0 0 0 0 0 ... 0 0 0 0 0 0 0 0 0 0
2022-12 0 0 0 0 0 0 0 0 0 0 ... 0 0 0 0 0 0 0 0 0 0
2023-01 0 0 0 0 0 0 0 0 0 0 ... 0 0 0 0 0 0 0 0 0 0
2023-02 0 0 0 0 0 0 0 0 0 0 ... 0 0 0 0 0 0 0 0 0 0
2023-03 0 0 0 0 0 0 0 0 0 0 ... 0 0 0 0 0 0 0 0 0 0

147 rows × 28 columns

In [5]:
df.index
Out [5]:
PeriodIndex(['2011-01', '2011-02', '2011-03', '2011-04', '2011-05', '2011-06',
             '2011-07', '2011-08', '2011-09', '2011-10',
             ...
             '2022-06', '2022-07', '2022-08', '2022-09', '2022-10', '2022-11',
             '2022-12', '2023-01', '2023-02', '2023-03'],
            dtype='period[M]', length=147)
In [6]:
# count words usage
from datetime import datetime
import re

# path = tweet_df.loc[0, 'path']
# print('path: ', path)
for path in tweet_df['path']:
    # determine period
    ts = datetime.strptime(path.stem, '%Y-%m-%d-%H-%M')
    ts = pd.to_datetime(ts).to_period('M')
    # print('period: ', ts)
    period_mask = ts == df.index
    # load in tweet
    with open(path, 'r') as f:
        text = f.read()
    # extract tweet text
    text = re.search(
        pattern='(?<=text: )(.|\n)*(?=images: )',
        string=text
    ).group(0)
    # determine if words present in text
    for word_root in df.columns[1:]:
        # print('word root: ', word_root)
        word_root_mask = word_root_df['root'].eq(word_root)
        word_matches = word_root_df.loc[word_root_mask, 'word_matches'].to_list()
        word_matches = sum(word_matches, []) # unpack nested list
        # print('word matches: ', word_matches)
        for word in word_matches:
            # print('word: ', word)
            if word in text:
                df.loc[period_mask, word_root] += 1

df
Out [6]:
root challeng SDG ESG recycling reduc planet CSR sustainab clean methanol ... environment green sulphur emissions zero climate eco mission CO2 bio
2011-01 0 0 0 0 0 0 0 0 0 0 ... 0 0 0 0 0 0 0 0 0 0
2011-02 0 0 0 0 0 0 0 0 0 0 ... 0 0 0 0 0 0 0 0 0 0
2011-03 0 0 0 0 0 0 0 0 0 0 ... 0 0 0 0 0 0 0 0 0 0
2011-04 0 0 0 0 0 0 0 0 0 0 ... 0 0 0 0 0 0 0 0 0 0
2011-05 0 0 0 0 0 0 0 0 0 0 ... 0 0 0 0 0 0 0 0 0 0
... ... ... ... ... ... ... ... ... ... ... ... ... ... ... ... ... ... ... ... ... ...
2022-11 0 0 0 5 0 0 0 5 0 1 ... 0 5 0 2 2 5 3 7 0 0
2022-12 0 0 0 0 2 0 0 6 4 3 ... 1 3 0 3 0 2 12 11 0 0
2023-01 0 0 1 0 2 0 0 3 0 0 ... 0 1 0 2 0 0 3 6 0 0
2023-02 0 0 1 0 0 0 0 4 4 1 ... 5 1 0 1 4 0 5 4 0 2
2023-03 0 0 0 1 1 0 0 4 2 1 ... 3 5 0 1 0 0 4 3 0 0

147 rows × 28 columns

In [7]:
# plot of combined use of root words over time
import matplotlib.pyplot as plt
import matplotlib.dates as mdates
import seaborn as sns
%matplotlib widget
plt.figure(figsize=(12,8))

time = df.index.to_timestamp()
counts = df.sum(axis=1)
plt.plot(time, counts, label='sustainability-related words')

mean = df.sum(axis=1).rolling(12, min_periods=1).mean()
plt.plot(time, mean, label='12-month mean')
std = df.sum(axis=1).rolling(12, min_periods=1).std()
plt.fill_between(
    x=time,
    y1=mean-std,
    y2=mean+std,
    alpha=0.2,
    label='12-month standard deviation'
)

locator = mdates.MonthLocator(interval=12)
fmt = mdates.DateFormatter('%Y-%m')
x_axis = plt.gca().xaxis
x_axis.set_major_locator(locator)
x_axis.set_major_formatter(fmt)
plt.xticks(rotation=45)
plt.ylabel('count (n)')
plt.legend(loc='upper left')
plt.tight_layout()
sns.despine(left=True)
sns.set_style("ticks",{'axes.grid' : True})
plt.savefig('plots/total_keyword_counts.png')
plt.show()

Above mean + std

In [8]:
# threshold = mean + std
# threshold_mask = counts > threshold
# threshold_mask

# tracked_words_list = list()
# for period in df.index[threshold_mask]:
#     # determine words with counts above mean + std
#     x = df.loc[period, :]
#     x = x[x>0] # ignore words with 0-count
#     threshold = x.mean() + x.std()
#     mask = df.loc[period, :].gt(threshold)
#     label_list = df.columns[mask].tolist()
#     for label in label_list:
#         tracked_words_list.append(label)
# tracked_words_list = list(set(tracked_words_list))
# tracked_words_list
In [9]:
# import seaborn as sns

# #define Seaborn color palette to use
# colors = sns.color_palette('Spectral', n_colors=len(tracked_words_list))
# plt.figure(figsize=(12, 8))

# for i, word in enumerate(tracked_words_list):
#     y = df.loc[:, word]
#     plt.plot(time, y, color=colors[i], label=word, alpha=0.8, linewidth=2)

# # ax = df.loc[:, tracked_words_list].plot(kind='line', layout='tight', figsize=(12,8))

# plt.ylabel('counts (n)')
# plt.xlabel('period')
# plt.legend()
# plt.show()

Below mean - std

In [10]:
# threshold = mean - std
# threshold_mask = counts < threshold
# threshold_mask

# tracked_words_list = list()
# for period in df.index[threshold_mask]:
#     # determine words with counts above mean + std
#     x = df.loc[period, :]
#     x = x[x>0] # ignore words with 0-count
#     threshold = x.mean() + x.std()
#     mask = df.loc[period, :].gt(threshold)
#     label_list = df.columns[mask].tolist()
#     for label in label_list:
#         tracked_words_list.append(label)
# tracked_words_list = list(set(tracked_words_list))
# tracked_words_list

Chosen Intervals

In [11]:
# period_list = [
#     '2015-12',
#     '2016-11',
#     '2018-08',
#     '2018-09',
#     '2021-03',
#     '2021-06',
#     '2021-08',
#     '2022-12'
# ]

# tracked_words_list = list()
# for period in period_list:
#     # determine words with counts above mean + std
#     x = df.loc[period, :]
#     x = x[x>0] # ignore words with 0-count
#     threshold = x.mean() + x.std()
#     mask = df.loc[period, :].gt(threshold)
#     label_list = df.columns[mask].tolist()
#     for label in label_list:
#         tracked_words_list.append(label)
# tracked_words_list = list(set(tracked_words_list))

# print(tracked_words_list)

# import seaborn as sns

# #define Seaborn color palette to use
# colors = sns.color_palette('Spectral', n_colors=len(tracked_words_list))
# plt.figure(figsize=(12, 8))

# for i, word in enumerate(tracked_words_list):
#     y = df.loc[:, word]
#     plt.plot(time, y, color=colors[i], label=word, alpha=0.8, linewidth=2)

# # ax = df.loc[:, tracked_words_list].plot(kind='line', layout='tight', figsize=(12,8))

# plt.ylabel('counts (n)')
# plt.xlabel('period')
# plt.legend()
# plt.show()

Piecharts of distribution from all periods

In [12]:
from tqdm.notebook import tqdm
for period in tqdm(df.index):
    x = df.loc[period, :]
    mask = x > 0
    x = x[mask]
    labels = df.columns[mask]

    #define Seaborn color palette to use
    colors = sns.color_palette('Blues')[0:len(df)]

    #create pie chart
    fig = plt.figure(figsize=(6,6))
    plt.pie(x=x, labels=labels, colors = colors, autopct='%.0f%%', startangle=0)
    plt.tight_layout()
    # plt.title(period)
    plt.savefig(f'word_distribution_pie_charts/total_keywords_{period}.png')
    plt.close()
    # plt.show()
  0%|          | 0/147 [00:00<?, ?it/s]

Determine Importance of individual words

In [ ]: