Období her -------------------------------------------------- games_metadata = [ { 'name': 'Team Fortress 2', 'path': 'data/440False.json', 'ref': ('2024-05-01', '2024-05-31'), 'rb': ('2024-06-01', '2024-06-30'), 'af': ('2024-07-01', '2024-07-31') }, { 'name': 'Apex Legends', 'path': 'data/1172470False.json' 'ref': ('2024-06-01', '2024-06-30'), 'rb': ('2024-07-01', '2024-07-31'), 'af': ('2024-08-01', '2024-08-31') }, { 'name': 'Skullgirls 2nd Encore', 'path': 'data/245170False.json', 'ref': ('2023-05-25', '2023-06-25'), 'rb': ('2023-06-26', '2023-07-15'), 'af': ('2023-07-16', '2023-08-16') }, { 'name': 'Tabletop Simulator', 'path': 'data/286160False.json', 'ref': ('2021-12-07', '2022-01-07'), 'rb': ('2022-01-08', '2022-02-06'), 'af': ('2022-02-07', '2022-03-07') }, { 'name': 'Resident Evil Revelations', 'path': 'data/222480False.json', 'ref': ('2023-12-01', '2023-12-31'), 'rb': ('2024-01-01', '2024-01-31'), 'af': ('2024-02-01', '2024-02-29') } ] Sběr dat -------------------------------------------------- import datetime import time import requests, json params = { 'json':1, # set the return result format to be json. 'language': 'english', 'cursor': '*', # set the cursor to retrieve reviews from a specific "page" 'num_per_page': 100, # retrieve more comments per request to reduce number of API calls 'filter': 'recent', 'filter_offtopic_activity':0, 'purchase_type' : 'all', 'review_type' : 'all' } gameid = 440 use_date = False date_from = time.mktime(datetime.datetime(2023, 5, 1, 0, 0).timetuple()) date_to = time.mktime(datetime.datetime(2023, 7, 31, 0, 0).timetuple()) seznam = [] while True: while True: try: resp = requests.get('https://store.steampowered.com/appreviews/' + str(gameid), params=params) break except Exception as e: print(e) time.sleep(1) try: data = resp.json() except Exception as e: print(e) break print(params) if data["success"] != 1 or len(data["reviews"]) == 0: print(data) print("End of data") break for review in data['reviews']: if not use_date or date_from <= review['timestamp_created'] <= date_to: seznam.append(review) else: print("Outside of time range: ", review['timestamp_created']) print("Number of reviews:", len(seznam)) params["cursor"] = data["cursor"] print(len(data["reviews"]),params["cursor"] ) #if data["reviews"][0]['timestamp_created'] < date_from: # break fn = "./data/" + str(gameid) + str(use_date) + ".json" with open(fn, "w") as f: json.dump(seznam, f) Náhled -------------------------------------------------- import json import pandas as pd import numpy as np data = json.load(open("data\\553850False.json", 'r')) data = pd.DataFrame.from_dict(data) data.info() data Analýza -------------------------------------------------- import pandas as pd GAMES_TO_ANALYZE = [ ("Skullgirls 2nd Encore", 'data/245170False.json'), # ..... ] REF_WINDOW = ('2023-05-25', '2023-06-25') CRIT_WINDOW = ('2023-06-26', '2023-07-15') def get_stats(dataframe, start, end): # Filtr na období mask = (dataframe['date_created'] >= start) & (dataframe['date_created'] <= end) p_df = dataframe.loc[mask] pos_df = p_df[p_df['voted_up'] == True].copy() if len(pos_df) == 0: return 0, 0 texts = pos_df['review'].fillna('').astype(str) char_counts = texts.str.len() word_counts = texts.apply(lambda x: len(x.split())) med_len = char_counts.median() is_nonstd = (char_counts < 3) | (word_counts == 1) nonstd_pct = (is_nonstd.sum() / len(pos_df)) * 100 return med_len, nonstd_pct all_results = [] for game_name, file_path in GAMES_TO_ANALYZE: try: df = pd.read_json(file_path) df['date_created'] = pd.to_datetime(df['timestamp_created'], unit='s') ref_med, ref_nonstd = get_stats(df, REF_WINDOW[0], REF_WINDOW[1]) crit_med, crit_nonstd = get_stats(df, CRIT_WINDOW[0], CRIT_WINDOW[1]) all_results.append({ "Hra": game_name, "Ref_Med": int(ref_med), "Ref_Nonstd": f"{ref_nonstd:.2f} %", "Crit_Med": int(crit_med), "Crit_Nonstd": f"{crit_nonstd:.2f} %" }) except Exception as e: print(f"Chyba u hry {game_name}: {e}") final_df = pd.DataFrame(all_results) latex_header = """ \\begin{table}[ht] \\centering \\caption{Analýza pozitivních recenzí: Medián délky a podíl nestandardních textů} \\label{tab:positive_review_analysis} \\begin{tabular}{l|rr|rr} \\hline & \\multicolumn{2}{c|}{\\textbf{Referenční období}} & \\multicolumn{2}{c}{\\textbf{Krizové období}} \\\\ \\textbf{Hra} & \\textit{Medián} & \\textit{Nestand.} & \\textit{Medián} & \\textit{Nestand.} \\\\ \\hline """ latex_rows = "" for _, row in final_df.iterrows(): latex_rows += f"{row['Hra']} & {row['Ref_Med']} & {row['Ref_Nonstd']} & {row['Crit_Med']} & {row['Crit_Nonstd']} \\\\ \n" latex_footer = """\\hline \\end{tabular} \\end{{table}} """ print(latex_header + latex_rows + latex_footer) -------------------------------------------------- import pandas as pd import re from collections import Counter FILE_PATH = 'data/222480False.json' REF_START, REF_END = '2023-12-01', '2023-12-31' KRIZE_START, KRIZE_END = '2024-01-01', '2024-01-31' STOPWORDS = { 'the', 'and', 'to', 'is', 'in', 'it', 'of', 'for', 'this', 'that', 'with', 'are', 'was', 'but', 'have', 'they', 'you', 'not', 'will', 'has', 'been', 'your', 'just', 'about', 'from', 'get', 'very', 'more', 'all', 'can', 'if', 'as', 'my', 'me', 'his', 'her', 'their', 'our', 'an', 'at', 'by', 'do', 'so', 'up', 'out', 'on', 'or', 'be', 'am', 'too', 'like', 'there', 'https', 'http', 'com', 'www', 'url', 'youtu', 'youtube' } KEYWORDS_TO_KEEP = {'fix', 'tf2', 'bot', 'valve'} STOPWORDS = STOPWORDS - KEYWORDS_TO_KEEP def get_word_stats(df, start_date, end_date): mask = (df['date_created'] >= start_date) & (df['date_created'] <= end_date) & (df['voted_up'] == False) reviews = df[mask]['review'].dropna().astype(str) words_list = [] for text in reviews: tokens = re.findall(r'#?\w+', text.lower()) filtered = [w for w in tokens if w not in STOPWORDS and (len(w) > 2 or w in KEYWORDS_TO_KEEP)] words_list.extend(filtered) return Counter(words_list), len(words_list) try: df = pd.read_json(FILE_PATH) df['date_created'] = pd.to_datetime(df['timestamp_created'], unit='s') counts_ref, total_ref = get_word_stats(df, REF_START, REF_END) counts_krize, total_krize = get_word_stats(df, KRIZE_START, KRIZE_END) all_crisis_words = counts_krize.keys() growth_list = [] for word in all_crisis_words: f_ref = (counts_ref.get(word, 0) / total_ref * 100) if total_ref > 0 else 0 f_krize = (counts_krize[word] / total_krize * 100) if total_krize > 0 else 0 diff = f_krize - f_ref growth_list.append((word, f_ref, f_krize, diff)) growth_list.sort(key=lambda x: x[3], reverse=True) top_15_growth = growth_list[:15] latex_code = [ "\\begin{table}[h!]", "\\centering", "\\small", "\\caption{TOP 15 slov s největším nárůstem relativní frekvence (v procentních bodech)}", "\\label{tab:rel_growth_top15}", "\\setlength{\\tabcolsep}{6pt}", "\\begin{tabular}{lrrr}", "\\hline", "\\textbf{Slovo} & \\textbf{Před (\\%)} & \\textbf{Během (\\%)} & \\textbf{Nárůst (p. b.)} \\\\ \\hline" ] for word, f_ref, f_krize, diff in top_15_growth: safe_word = word.replace("#", "\\#") latex_code.append( f"{safe_word} & {f_ref:.2f} & {f_krize:.2f} & {diff:+.2f} \\\\" ) latex_code.append("\\hline") latex_code.append("\\end{tabular}") latex_code.append("\\end{table}") print("\n".join(latex_code)) except Exception as e: print(f"Chyba: {e}") -------------------------------------------------- import pandas as pd FILE_PATH = 'data/245170False.json' KRIZE_START = '2023-06-26' KRIZE_END = '2023-07-15' SEARCH_WORD = 'good' try: df = pd.read_json(FILE_PATH) df['date_created'] = pd.to_datetime(df['timestamp_created'], unit='s') mask = (df['date_created'] >= KRIZE_START) & \ (df['date_created'] <= KRIZE_END) & \ (df['voted_up'] == False) df_krize = df[mask].copy() pattern = rf'\b{SEARCH_WORD}\b' df_love = df_krize[df_krize['review'].str.contains(pattern, case=False, na=False)] if len(df_love) > 0: n_sample = min(len(df_love), 5) samples = df_love.sample(n=n_sample, random_state=42) print(f"--- Nalezeno celkem {len(df_love)} negativních recenzí se slovem '{SEARCH_WORD}' ---") print(f"Vypisuji {n_sample} náhodných ukázek:\n") for i, (idx, row) in enumerate(samples.iterrows(), 1): date_str = row['date_created'].strftime('%Y-%m-%d') print(f"VZOREK č. {i} (Datum: {date_str})") print("-" * 50) print(row['review']) print("-" * 50 + "\n") else: print(f"V krizovém období nebyly nalezeny žádné negativní recenze obsahující slovo '{SEARCH_WORD}'.") except Exception as e: print(f"Chyba při zpracování: {e}") -------------------------------------------------- import pandas as pd import re import numpy as np from collections import Counter from wordcloud import WordCloud import matplotlib.pyplot as plt FILE_PATH = 'data/222480False.json' REF_START, REF_END = '2023-12-01', '2023-12-31' KRIZE_START, KRIZE_END = '2024-01-01', '2024-01-31' STOPWORDS = { 'the', 'and', 'to', 'is', 'in', 'it', 'of', 'for', 'this', 'that', 'with', 'are', 'was', 'but', 'have', 'they', 'you', 'not', 'will', 'has', 'been', 'your', 'just', 'about', 'from', 'get', 'very', 'more', 'all', 'can', 'if', 'as', 'my', 'me', 'his', 'her', 'their', 'our', 'an', 'at', 'by', 'do', 'so', 'up', 'out', 'on', 'or', 'be', 'am', 'too', 'like', 'there', 'even', 'game', 'play', 'which', 'did', 'could' } COLOR_REF = '#5c88da' COLOR_KRIZE = '#e07a5f' def get_clean_stats(df, start, end): """Filtruje negativní recenze a vrací počty slov + celkové N.""" mask = (df['date_created'] >= start) & \ (df['date_created'] <= end) & \ (df['voted_up'] == False) reviews = df[mask]['review'].dropna().astype(str) total_tokens = 0 filtered_list = [] for text in reviews: tokens = re.findall(r'#?\w+', text.lower()) total_tokens += len(tokens) keywords = [w for w in tokens if w not in STOPWORDS and len(w) > 2] filtered_list.extend(keywords) return Counter(filtered_list), total_tokens try: # Načtení dat df = pd.read_json(FILE_PATH) df['date_created'] = pd.to_datetime(df['timestamp_created'], unit='s') counts_ref, n_ref = get_clean_stats(df, REF_START, REF_END) counts_krize, n_krize = get_clean_stats(df, KRIZE_START, KRIZE_END) all_words = set(counts_krize.keys()) | set(counts_ref.keys()) word_frequencies = {} word_colors = {} for word in all_words: f_ref = (counts_ref.get(word, 0) / n_ref) * 100 if n_ref > 0 else 0 f_krize = (counts_krize.get(word, 0) / n_krize) * 100 if n_krize > 0 else 0 word_frequencies[word] = max(f_ref, f_krize) word_colors[word] = COLOR_KRIZE if f_krize > f_ref else COLOR_REF x, y = np.ogrid[:1000, :1000] mask = (x - 500) ** 2 + (y - 500) ** 2 > 400 ** 2 mask = 255 * mask.astype(int) wc = WordCloud( background_color=None, mode="RGBA", mask=mask, max_words=70, relative_scaling=0.4, width=1000, height=1000, prefer_horizontal=0.9 ).generate_from_frequencies(word_frequencies) # 3. Barevná funkce def color_func(word, **kwargs): return word_colors.get(word, '#888888') fig = plt.figure(figsize=(10, 10)) ax = plt.Axes(fig, [0., 0., 1., 1.]) ax.set_axis_off() fig.add_axes(ax) plt.imshow(wc.recolor(color_func=color_func), interpolation="bilinear") plt.text(0.5, 0.05, f'Referenční období (Pouze negativní recenze, N={n_ref} slov)', color=COLOR_REF, fontsize=12, fontweight='bold', ha='center', transform=plt.gca().transAxes) plt.text(0.5, 0.02, f'Období review bombingu (Pouze negativní recenze během RB, N={n_krize} slov)', color=COLOR_KRIZE, fontsize=12, fontweight='bold', ha='center', transform=plt.gca().transAxes) plt.savefig( 'wordcloud_res_final.png', transparent=True, bbox_inches='tight', pad_inches=0, dpi=300 ) except Exception as e: print(f"Chyba při zpracování: {e}") -------------------------------------------------- import pandas as pd import seaborn as sns import matplotlib.pyplot as plt games_metadata = [ { 'name': 'Team Fortress 2', 'path': 'data/440False.json', 'ref': ('2024-05-01', '2024-05-31'), 'rb': ('2024-06-01', '2024-06-30') }, { 'name': 'Apex Legends', 'path': 'data/1172470False.json', 'ref': ('2024-06-01', '2024-06-30'), 'rb': ('2024-07-01', '2024-07-31') }, { 'name': 'Skullgirls 2nd Encore', 'path': 'data/245170False.json', 'ref': ('2023-05-25', '2023-06-25'), 'rb': ('2023-06-26', '2023-07-15') }, { 'name': 'Tabletop Simulator', 'path': 'data/286160False.json', 'ref': ('2021-12-07', '2022-01-07'), 'rb': ('2022-01-08', '2022-02-06') }, { 'name': 'Resident Evil Revelations', 'path': 'data/222480False.json', 'ref': ('2023-12-01', '2023-12-31'), 'rb': ('2024-01-01', '2024-01-31') } ] all_rb_data = [] for game in games_metadata: df = pd.read_json(game['path']) df['date'] = pd.to_datetime(df['timestamp_created'], unit='s') df['hours'] = df['author'].apply(lambda x: x.get('playtime_forever', 0)) / 60 ref_mask = (df['date'] >= game['ref'][0]) & (df['date'] <= game['ref'][1]) & (df['voted_up'] == False) median_ref = df[ref_mask]['hours'].median() if median_ref == 0 or pd.isna(median_ref): median_ref = 1.0 rb_mask = (df['date'] >= game['rb'][0]) & (df['date'] <= game['rb'][1]) rb_df = df[rb_mask].copy() rb_df['Game'] = game['name'] rb_df['playtime_index'] = rb_df['hours'] / median_ref rb_df['Sentiment'] = rb_df['voted_up'].map({True: 'Pozitivní', False: 'Negativní'}) all_rb_data.append(rb_df) final_rb_df = pd.concat(all_rb_data) plt.figure(figsize=(15, 8)) plt.style.use('seaborn-v0_8-whitegrid') sns.boxplot( data=final_rb_df[final_rb_df['playtime_index'] <= 10], x='Game', y='playtime_index', hue='Sentiment', palette={'Pozitivní': '#90be6d', 'Negativní': '#e07a5f'}, fliersize=1.5 ) sns.set_context("paper", font_scale=1.5) plt.rcParams.update({'font.size': 30}) plt.axhline(1.0, color='black', linestyle='--', alpha=0.4, label='Běžná nespokojenost (Ref. medián)') plt.title('Srovnání herní zkušenosti útočníků a obránců', fontsize=25) plt.ylabel('Index herního času', fontsize=25) plt.xlabel('Analyzované hry', fontsize=25) plt.legend(title='Sentiment v době RB', fontsize=18) plt.xticks(fontsize=17, fontweight='bold') plt.tight_layout() plt.savefig('rb_only_comparison.png', dpi=300) plt.show() print(final_rb_df.groupby(['Game', 'Sentiment'])['playtime_index'].median().unstack()) -------------------------------------------------- import pandas as pd import seaborn as sns import matplotlib.pyplot as plt games_metadata = [ { 'name': 'Team Fortress 2', 'path': 'data/440False.json', 'ref': ('2024-05-01', '2024-05-31'), 'rb': ('2024-06-01', '2024-06-30') }, { 'name': 'Apex Legends', 'path': 'data/1172470False.json', 'ref': ('2024-06-01', '2024-06-30'), 'rb': ('2024-07-01', '2024-07-31') }, { 'name': 'Skullgirls 2nd Encore', 'path': 'data/245170False.json', 'ref': ('2023-05-25', '2023-06-25'), 'rb': ('2023-06-26', '2023-07-15') }, { 'name': 'Tabletop Simulator', 'path': 'data/286160False.json', 'ref': ('2021-12-07', '2022-01-07'), 'rb': ('2022-01-08', '2022-02-06') }, { 'name': 'Resident Evil Revelations', 'path': 'data/222480False.json', 'ref': ('2023-12-01', '2023-12-31'), 'rb': ('2024-01-01', '2024-01-31') } ] all_ref_data = [] for game in games_metadata: df = pd.read_json(game['path']) df['date'] = pd.to_datetime(df['timestamp_created'], unit='s') df['hours'] = df['author'].apply(lambda x: x.get('playtime_forever', 0)) / 60 ref_mask = (df['date'] >= game['ref'][0]) & (df['date'] <= game['ref'][1]) ref_df = df[ref_mask].copy() median_val = ref_df[ref_df['voted_up'] == False]['hours'].median() if median_val == 0 or pd.isna(median_val): median_val = 1.0 ref_df['Game'] = game['name'] ref_df['playtime_index'] = ref_df['hours'] / median_val ref_df['Sentiment'] = ref_df['voted_up'].map({True: 'Pozitivní', False: 'Negativní'}) all_ref_data.append(ref_df) final_ref_df = pd.concat(all_ref_data) plt.figure(figsize=(15, 8)) plt.style.use('seaborn-v0_8-whitegrid') sns.boxplot( data=final_ref_df[final_ref_df['playtime_index'] <= 10], x='Game', y='playtime_index', hue='Sentiment', palette={'Pozitivní': '#298c8c', 'Negativní': '#ffa600'}, fliersize=1.5 ) sns.set_context("paper", font_scale=1.5) plt.rcParams.update({'font.size': 30}) plt.axhline(1.0, color='black', linestyle='--', alpha=0.5, label='Standard nespokojenosti (1.0)') plt.title('Herní zkušenost autorů recenzí v referenčním období', fontsize=30) plt.ylabel('Index herního času', fontsize=25) plt.xlabel('Analyzované hry', fontsize=25) plt.legend(title='Sentiment (Před krizí)', fontsize=18) plt.xticks(fontsize=17, fontweight='bold') plt.tight_layout() plt.savefig('baseline_comparison.png', dpi=300) plt.show() print(final_ref_df.groupby(['Game', 'Sentiment'])['playtime_index'].median().unstack()) -------------------------------------------------- import pandas as pd import seaborn as sns import matplotlib.pyplot as plt games_metadata = [ { 'name': 'Team Fortress 2', 'path': 'data/440False.json', 'ref': ('2024-05-01', '2024-05-31'), 'rb': ('2024-06-01', '2024-06-30') }, { 'name': 'Apex Legends', 'path': 'data/1172470False.json', 'ref': ('2024-06-01', '2024-06-30'), 'rb': ('2024-07-01', '2024-07-31') }, { 'name': 'Skullgirls 2nd Encore', 'path': 'data/245170False.json', 'ref': ('2023-05-25', '2023-06-25'), 'rb': ('2023-06-26', '2023-07-15') }, { 'name': 'Tabletop Simulator', 'path': 'data/286160False.json', 'ref': ('2021-12-07', '2022-01-07'), 'rb': ('2022-01-08', '2022-02-06') }, { 'name': 'Resident Evil Revelations', 'path': 'data/222480False.json', 'ref': ('2023-12-01', '2023-12-31'), 'rb': ('2024-01-01', '2024-01-31') } ] all_processed_data = [] for game in games_metadata: print(f"Zpracovávám: {game['name']}...") # Načtení JSONu df = pd.read_json(game['path']) df['date'] = pd.to_datetime(df['timestamp_created'], unit='s') df['hours_forever'] = df['author'].apply(lambda x: x.get('playtime_forever', 0)) / 60 df['hours_at_review'] = df['author'].apply(lambda x: x.get('playtime_at_review', 0)) / 60 df['delta_hours'] = df['hours_forever'] - df['hours_at_review'] ref_mask = (df['date'] >= game['ref'][0]) & (df['date'] <= game['ref'][1]) rb_mask = (df['date'] >= game['rb'][0]) & (df['date'] <= game['rb'][1]) median_ref_neg = df[ref_mask & (df['voted_up'] == False)]['hours_forever'].median() if median_ref_neg == 0 or pd.isna(median_ref_neg): median_ref_neg = 1.0 period_df = df[ref_mask | rb_mask].copy() period_df['playtime_index'] = period_df['hours_forever'] / median_ref_neg period_df['Game'] = game['name'] period_df['Sentiment'] = period_df['voted_up'].map({True: 'Pozitivní', False: 'Negativní'}) period_df['Období'] = period_df['date'].apply(lambda x: 'Referenční' if x <= pd.to_datetime(game['ref'][1]) else 'Review Bombing') all_processed_data.append(period_df) final_df = pd.concat(all_processed_data) plt.rcParams.update({'font.size': 14, 'font.family': 'sans-serif'}) sns.set_style("whitegrid") plt.figure(figsize=(16, 9)) rb_only = final_df[final_df['Období'] == 'Review Bombing'] ax1 = sns.boxplot( data=rb_only[rb_only['playtime_index'] <= 10], x='Game', y='playtime_index', hue='Sentiment', palette={'Pozitivní': '#298c8c', 'Negativní': '#ffa600'}, fliersize=2, linewidth=1.5 ) plt.axhline(1.0, color='black', linestyle='--', alpha=0.5) plt.title('Srovnání herní zkušenosti útočníků a obránců během Review Bombingu', fontsize=20, pad=20, fontweight='bold') plt.ylabel('Playtime Index (1.0 = medián nespokojeného hráče)', fontsize=16) plt.xlabel('', fontsize=16) plt.xticks(fontsize=15, fontweight='bold') plt.legend(title='Sentiment', fontsize=14, title_fontsize=14) plt.tight_layout() plt.savefig('graf_rb_final.pdf', bbox_inches='tight') plt.show() plt.figure(figsize=(16, 9)) ax2 = sns.boxplot( data=rb_only[rb_only['delta_hours'] <= 50], x='Game', y='delta_hours', hue='Sentiment', palette={'Pozitivní': '#298c8c', 'Negativní': '#ffa600'} ) plt.title('Kolik hodin uživatelé nahráli PO napsání recenze v době RB', fontsize=20, pad=20, fontweight='bold') plt.ylabel('Počet hodin nahraných po recenzi', fontsize=25) plt.xlabel('', fontsize=25) plt.xticks(fontsize=17, fontweight='bold') plt.tight_layout() plt.savefig('graf_delta_playtime.pdf', bbox_inches='tight') plt.show() print("\n--- MEDIÁNY PLAYTIME INDEXU V DOBĚ RB ---") stats = rb_only.groupby(['Game', 'Sentiment'])['playtime_index'].median().unstack() print(stats) print("\n--- MEDIÁNY DELTA PLAYTIME (HODINY PO RECENZI) ---") delta_stats = rb_only.groupby(['Game', 'Sentiment'])['delta_hours'].median().unstack() print(delta_stats) -------------------------------------------------- import pandas as pd import seaborn as sns import matplotlib.pyplot as plt games_metadata = [ { 'name': 'Team Fortress 2', 'path': 'data/440False.json', 'ref': ('2024-05-01', '2024-05-31'), 'rb': ('2024-06-01', '2024-06-30') }, { 'name': 'Apex Legends', 'path': 'data/1172470False.json', 'ref': ('2024-06-01', '2024-06-30'), 'rb': ('2024-07-01', '2024-07-31') }, { 'name': 'Skullgirls 2nd Encore', 'path': 'data/245170False.json', 'ref': ('2023-05-25', '2023-06-25'), 'rb': ('2023-06-26', '2023-07-15') }, { 'name': 'Tabletop Simulator', 'path': 'data/286160False.json', 'ref': ('2021-12-07', '2022-01-07'), 'rb': ('2022-01-08', '2022-02-06') }, { 'name': 'Resident Evil Revelations', 'path': 'data/222480False.json', 'ref': ('2023-12-01', '2023-12-31'), 'rb': ('2024-01-01', '2024-01-31') } ] helpful_analysis = [] for game in games_metadata: df = pd.read_json(game['path']) df['date'] = pd.to_datetime(df['timestamp_created'], unit='s') ref_mask = (df['date'] >= game['ref'][0]) & (df['date'] <= game['ref'][1]) rb_mask = (df['date'] >= game['rb'][0]) & (df['date'] <= game['rb'][1]) temp_df = df[ref_mask | rb_mask].copy() temp_df['Game'] = game['name'] temp_df['Sentiment'] = temp_df['voted_up'].map({True: 'Pozitivní', False: 'Negativní'}) temp_df['Období'] = temp_df['date'].apply(lambda x: 'Referenční' if x <= pd.to_datetime(game['ref'][1]) else 'Review Bombing') helpful_analysis.append(temp_df[['Game', 'Sentiment', 'Období', 'votes_up', 'weighted_vote_score']]) final_helpful_df = pd.concat(helpful_analysis) summary = final_helpful_df.groupby(['Game', 'Období', 'Sentiment'])['votes_up'].mean().unstack() plt.figure(figsize=(16, 8)) sns.set_style("whitegrid") rb_data = final_helpful_df[final_helpful_df['Období'] == 'Review Bombing'] ax = sns.barplot( data=rb_data, x='Game', y='votes_up', hue='Sentiment', palette={'Pozitivní': '#298c8c', 'Negativní': '#ffa600'}, errorbar=None ) plt.title('Průměrný počet "Helpful" hlasů na jednu recenzi (Období RB)', fontsize=18, fontweight='bold') plt.ylabel('Průměrný počet hlasů', fontsize=14) plt.xlabel('', fontsize=14) plt.xticks(fontsize=16, fontweight='bold') for p in ax.patches: height = p.get_height() if height > 0.1: ax.annotate(format(p.get_height(), '.1f'), (p.get_x() + p.get_width() / 2., p.get_height()), ha = 'center', va = 'center', xytext = (0, 9), textcoords = 'offset points', fontsize=12) plt.tight_layout() plt.savefig('helpful_votes_comparison.pdf') plt.show() print("\n--- INDEX VIDITELNOSTI (Kolikrát víc hlasů má Negativní vs. Pozitivní) ---") visibility_index = summary['Negativní'] / summary['Pozitivní'] print(visibility_index.unstack()) -------------------------------------------------- correlation_results = [] for game in games_metadata: df = pd.read_json(game['path']) author_df = pd.json_normalize(df['author']) df = pd.concat([df.drop(['author'], axis=1), author_df], axis=1) playtime_col = 'playtime_at_review' if 'playtime_at_review' in df.columns else 'playtime_forever' df['date'] = pd.to_datetime(df['timestamp_created'], unit='s') ref_mask = (df['date'] >= game['ref'][0]) & (df['date'] <= game['ref'][1]) rb_mask = (df['date'] >= game['rb'][0]) & (df['date'] <= game['rb'][1]) for period_name, mask in [('Referenční', ref_mask), ('Review Bombing', rb_mask)]: subset = df[mask].copy() if len(subset) > 5: corr, p_value = spearmanr(subset[playtime_col], subset['votes_up'], nan_policy='omit') correlation_results.append({ 'Hra': game['name'], 'Období': period_name, 'Spearman koeficient': round(corr, 3), 'p-hodnota': p_value, 'Vzorek': len(subset) }) corr_df = pd.DataFrame(correlation_results) print(corr_df[['Hra', 'Období', 'Spearman koeficient', 'p-hodnota']]) -------------------------------------------------- import pandas as pd import json import matplotlib.pyplot as plt import seaborn as sns from datetime import datetime rb_configs = [ {'name': 'Team Fortress 2', 'path': 'data/440False.json', 'ref': ('2024-05-01', '2024-05-31'), 'rb': ('2024-06-01', '2024-06-30')}, {'name': 'Apex Legends', 'path': 'data/1172470False.json', 'ref': ('2024-06-01', '2024-06-30'), 'rb': ('2024-07-01', '2024-07-31')}, {'name': 'Skullgirls 2nd Encore', 'path': 'data/245170False.json', 'ref': ('2023-05-25', '2023-06-25'), 'rb': ('2023-06-26', '2023-07-15')}, {'name': 'Tabletop Simulator', 'path': 'data/286160False.json', 'ref': ('2021-12-07', '2022-01-07'), 'rb': ('2022-01-08', '2022-02-06')}, {'name': 'Resident Evil Revelations', 'path': 'data/222480False.json', 'ref': ('2023-12-01', '2023-12-31'), 'rb': ('2024-01-01', '2024-01-31')} ] all_data = [] for config in rb_configs: with open(config['path'], 'r', encoding='utf-8') as f: data = json.load(f) df = pd.DataFrame(data if isinstance(data, list) else data['reviews']) df['date'] = pd.to_datetime(df['timestamp_created'], unit='s').dt.date daily = df.groupby('date').size().reset_index(name='count') daily['date'] = pd.to_datetime(daily['date']) ref_start, ref_end = pd.to_datetime(config['ref'][0]), pd.to_datetime(config['ref'][1]) baseline = daily[(daily['date'] >= ref_start) & (daily['date'] <= ref_end)]['count'].mean() if baseline <= 0 or pd.isna(baseline): baseline = 1 rb_start = pd.to_datetime(config['rb'][0]) daily['rel_day'] = (daily['date'] - rb_start).dt.days daily['growth_index'] = daily['count'] / baseline subset = daily[(daily['rel_day'] >= -30) & (daily['rel_day'] <= 60)].copy() subset['game'] = config['name'] all_data.append(subset) final_df = pd.concat(all_data) plt.figure(figsize=(10, 7)) sns.set_style("whitegrid") ax = sns.lineplot(data=final_df, x='rel_day', y='growth_index', hue='game', linewidth=2.5) plt.axvline(0, color='black', linestyle='--', linewidth=1.5, label='Start kontroverze (Den 0)') plt.axvline(30, color='gray', linestyle=':', linewidth=1.2, label='Konec 30denní vlny') plt.axvspan(-30, 0, color='gray', alpha=0.1, label='Fáze: Před (Normál)') plt.axvspan(0, 30, color='red', alpha=0.05, label='Fáze: Během (Krize)') plt.axvspan(30, 60, color='blue', alpha=0.05, label='Fáze: Po (Dozvuk)') plt.title('Srovnání dynamiky produkce recenzí ve třech fázích', fontsize=15, pad=20) plt.xlabel('Relativní dny (Day 0 = Začátek RB)', fontsize=12) plt.ylabel('Index nárůstu (x-krát více oproti normálu)', fontsize=12) plt.legend(title='Titul', loc='upper center', bbox_to_anchor=(0.5, -0.15), ncol=3, frameon=True) plt.tight_layout() plt.savefig('graf_final.png', dpi=300, bbox_inches='tight') plt.show() -------------------------------------------------- import pandas as pd import json games_metadata = [ { 'name': 'Team Fortress 2', 'path': 'data/440False.json', 'ref': ('2024-05-01', '2024-05-31'), 'rb': ('2024-06-01', '2024-06-30'), 'af': ('2024-07-01', '2024-07-31') }, { 'name': 'Apex Legends', 'path': 'data/1172470False.json', 'ref': ('2024-06-01', '2024-06-30'), 'rb': ('2024-07-01', '2024-07-31'), 'af': ('2024-08-01', '2024-08-31') }, { 'name': 'Skullgirls 2nd Encore', 'path': 'data/245170False.json', 'ref': ('2023-05-25', '2023-06-25'), 'rb': ('2023-06-26', '2023-07-15'), 'af': ('2023-07-16', '2023-08-16') }, { 'name': 'Tabletop Simulator', 'path': 'data/286160False.json', 'ref': ('2021-12-07', '2022-01-07'), 'rb': ('2022-01-08', '2022-02-06'), 'af': ('2022-02-07', '2022-03-07') }, { 'name': 'Resident Evil Revelations', 'path': 'data/222480False.json', 'ref': ('2023-12-01', '2023-12-31'), 'rb': ('2024-01-01', '2024-01-31'), 'af': ('2024-02-01', '2024-02-29') } ] def get_stats(df, range_tuple): if not range_tuple: return 0, 0 start, end = pd.to_datetime(range_tuple[0]), pd.to_datetime(range_tuple[1]) mask = (df['date'] >= start) & (df['date'] <= end) period_df = df.loc[mask] total = len(period_df) if total == 0: return 0, 0 neg_count = len(period_df[period_df['voted_up'] == False]) neg_pct = (neg_count / total) * 100 return total, neg_pct rows = [] for game in games_metadata: try: with open(game['path'], 'r', encoding='utf-8') as f: data = json.load(f) df = pd.DataFrame(data if isinstance(data, list) else data['reviews']) df['date'] = pd.to_datetime(df['timestamp_created'], unit='s') n_ref, p_ref = get_stats(df, game.get('ref')) n_rb, p_rb = get_stats(df, game.get('rb')) n_af, p_af = get_stats(df, game.get('af')) f_p_ref = str(round(p_ref, 1)).replace('.', ',') f_p_rb = str(round(p_rb, 1)).replace('.', ',') f_p_af = str(round(p_af, 1)).replace('.', ',') rows.append(f"{game['name']} & {n_ref} & {f_p_ref}\\% & {n_rb} & {f_p_rb}\\% & {n_af} & {f_p_af}\\% \\\\") except Exception as e: print(f"Chyba u hry {game['name']}: {e}") latex_output = r""" \begin{table}[H] \centering \caption{Srovnání objemu recenzí a negativity dle definovaných fází} \label{tab:final_stats} \footnotesize \setlength{\tabcolsep}{4pt} \begin{tabular}{l rr rr rr} \toprule & \multicolumn{2}{c}{Referenční (Před)} & \multicolumn{2}{c}{Review Bombing} & \multicolumn{2}{c}{Dozvuk (Po)} \\ \cmidrule(lr){2-3} \cmidrule(lr){4-5} \cmidrule(lr){6-7} Titul & $N$ & Neg. (\%) & $N$ & Neg. (\%) & $N$ & Neg. (\%) \\ \midrule """ latex_output += "\n".join(rows) latex_output += r""" \bottomrule \end{tabular} \end{table} """ print(latex_output) -------------------------------------------------- import pandas as pd import json games_metadata = [ {'name': 'Team Fortress 2', 'path': 'data/440False.json', 'rb': ('2024-06-01', '2024-06-30')}, {'name': 'Apex Legends', 'path': 'data/1172470False.json', 'rb': ('2024-07-01', '2024-07-31')}, {'name': 'Skullgirls 2nd Encore', 'path': 'data/245170False.json', 'rb': ('2023-06-26', '2023-07-15')}, {'name': 'Tabletop Simulator', 'path': 'data/286160False.json', 'rb': ('2022-01-08', '2022-02-06')}, {'name': 'Resident Evil Revelations', 'path': 'data/222480False.json', 'rb': ('2024-01-01', '2024-01-31')} ] metrics = [ ('Vlastněné hry', 'num_games_owned', 1), ('Počet recenzí', 'num_reviews', 1), ('Herní doba (h)', 'playtime_forever', 60) ] latex_rows = [] for game in games_metadata: with open(game['path'], 'r', encoding='utf-8') as f: data = json.load(f) df = pd.DataFrame(data if isinstance(data, list) else data['reviews']) df['date'] = pd.to_datetime(df['timestamp_created'], unit='s') start, end = pd.to_datetime(game['rb'][0]), pd.to_datetime(game['rb'][1]) mask = (df['date'] >= start) & (df['date'] <= end) & (df['voted_up'] == False) authors = pd.json_normalize(df.loc[mask, 'author']) authors = authors[authors['num_games_owned'] > 0] latex_rows.append(f"\\multicolumn{{4}}{{l}}{{\\textbf{{{game['name']}}}}} \\\\ \\hline") for label, col, div in metrics: vals = authors[col] / div if not vals.empty: med = f"{vals.median():.1f}".replace('.', ',') mi = f"{vals.min():.1f}".replace('.', ',') ma = f"{vals.max():.1f}".replace('.', ',') latex_rows.append(f"{label} & {med} & {mi} & {ma} \\\\") else: latex_rows.append(f"{label} & 0 & 0 & 0 \\\\") latex_rows.append("\\midrule") latex_table = r""" \begin{table}[H] \centering \caption{Statistický profil recenzentů v období RB (bez soukromých účtů)} \label{tab:reviewer_profiles_vertical} \begin{tabular}{l rrr} \toprule \textbf{Metrika} & \textbf{Medián} & \textbf{Min} & \textbf{Max} \\ \midrule """ latex_table += "\n".join(latex_rows) latex_table += r""" \bottomrule \end{tabular} \end{table} """ print(latex_table) -------------------------------------------------- -------------------------------------------------- --------------------------------------------------