Move .csvs to subdirectory, refactor cube_json.py out of word_count.py

This commit is contained in:
henriquenakashima 2020-11-29 15:24:37 -05:00
parent 7226fc3b17
commit 32f8c20b2e
8 changed files with 110 additions and 95 deletions

View file

@ -1,6 +1,8 @@
import json import json
import os import os
import cube_json
import word_count
def get_keywords(card): def get_keywords(card):
if 'keywords' in card.keys(): if 'keywords' in card.keys():
@ -40,36 +42,40 @@ def keyword_report(k_dict):
print(f'\nTotal unique keywords: {len(k_dict)}') print(f'\nTotal unique keywords: {len(k_dict)}')
# ### Example use: print a keyword report for a cards.json file ### if __name__ == '__main__':
# filename = f'cubes/CoreTheElegantCube.json' ### Example use: print a keyword report for a cards.json file ###
# k_dict = keyword_count(filename) # cube_list = cube_json.load_cube_from_csv('cube_csvs/TheElegantCube_2020-11-17_5.0.4.csv', {'core'})
# keyword_report(k_dict) # cube_json_filename = cube_json.create_cube_json(cube_list, 'cubes/TheElegantCube_2020-11-17_5.0.4.json')
# keyword_dict = keyword_count(cube_json_filename)
# keyword_report(keyword_dict)
# ### Example use: count keywords in a list of cube.json files ### # ### Example use: count keywords in a list of cube.json files ###
# output = open('cube_keyword_frequency.csv', 'w+') # output = open('cube_keyword_frequency.csv', 'w+')
# all_cube_files = os.listdir('cubes') # all_cube_files = os.listdir('cubes')
# output_file = open('cube_keyword_frequency.csv', 'w+') # output_file = open('cube_keyword_frequency.csv', 'w+')
# output_file.write('Cube,Keywords\n') # output_file.write('Cube,Keywords\n')
# for filename in all_cube_files: # for filename in all_cube_files:
# if filename.endswith('json'): # if filename.endswith('json'):
# k_dict = keyword_count('cubes/' + filename) # k_dict = keyword_count('cubes/' + filename)
# output_file.write(f"{filename.strip('json')},{len(k_dict)}\n") # output_file.write(f"{filename.strip('json')},{len(k_dict)}\n")
# print(filename, len(k_dict)) # print(filename, len(k_dict))
# output_file.close() # output_file.close()
# ### Example use: find keyword count in each set ### # ### Example use: find keyword count in each set ###
# f = open('set_codes.txt') # f = open('set_codes.txt')
# codes = [line.strip() for line in f.readlines()] # codes = [line.strip() for line in f.readlines()]
# output_file = open('set_keyword_frequency.csv', 'w+') # output_file = open('set_keyword_frequency.csv', 'w+')
# output_file.write('Set,Keywords\n') # output_file.write('Set,Keywords\n')
# for code in codes: # for code in codes:
# if code == 'CON': # if code == 'CON':
# code = 'CON_' # code = 'CON_'
# filename = f'scryfall/sets/{code}.json' # filename = f'scryfall/sets/{code}.json'
# k_dict = keyword_count(filename) # k_dict = keyword_count(filename)
# print(f'{code}: {len(k_dict)}') # print(f'{code}: {len(k_dict)}')
# output_file.write(f'{code},{len(k_dict)}\n') # output_file.write(f'{code},{len(k_dict)}\n')
# output_file.close() # output_file.close()
# f.close() # f.close()
pass

View file

Can't render this file because it has a wrong number of fields in line 2.

View file

Can't render this file because it has a wrong number of fields in line 2.

View file

Can't render this file because it has a wrong number of fields in line 2.

41
cube_json.py Normal file
View file

@ -0,0 +1,41 @@
import csv
import json
from word_count import EXPECTED_HEADER, COLUMNS, TEMP_JSON, is_actual_mtg_card
def load_cube_from_csv(filename, tag_filter=None):
cards = []
with open(filename) as f:
reader = csv.reader(f)
header_line = next(reader)
assert header_line == EXPECTED_HEADER
for line in reader:
card = line[COLUMNS['Name']]
# Export CSV is broken on CubeCobra because Image Back URL is always just one double quotes character.
tags = [t.strip() for t in line[COLUMNS['Tags']].split(', ')]
if tag_filter is None or any(t in tag_filter for t in tags):
cards.append(card)
return cards
def create_cube_json(cards, output_filename=TEMP_JSON):
# Call this once to create a smaller JSON file from the full card list
# Visit https://scryfall.com/docs/api/bulk-data and look for "Oracle Cards" file;
# Download and rename it to 'oracle_cards.json'
with open('oracle_cards.json', encoding='utf8') as f_oracle_cards:
d = json.load(f_oracle_cards)
f_cube_json = open(output_filename, 'w+', encoding='utf8')
cube_data = []
for card in d:
if card['name'] in cards:
for i in range(cards.count(card['name'])):
cube_data += [card]
elif 'card_faces' in card.keys() and is_actual_mtg_card(card):
for face in card['card_faces']:
if face['name'] in cards:
cube_data += [card]
string_data = json.dumps(cube_data)
f_cube_json.write(string_data)
f_cube_json.close()
return output_filename

File diff suppressed because one or more lines are too long

View file

@ -8,7 +8,7 @@ OUTPUT_POOL_SIZE = 360
CARDS_FROM_OCCASIONAL = 48 CARDS_FROM_OCCASIONAL = 48
CARDS_FROM_CORE = OUTPUT_POOL_SIZE - CARDS_FROM_OCCASIONAL CARDS_FROM_CORE = OUTPUT_POOL_SIZE - CARDS_FROM_OCCASIONAL
CSV_FILENAME = 'TheElegantCube20201102.csv' CSV_FILENAME = 'cube_csvs/TheElegantCube_2020-11-17_5.0.4.csv'
OUTPUT_FILENAME = 'cards_in_draft.txt' OUTPUT_FILENAME = 'cards_in_draft.txt'
EXPECTED_HEADER = [ EXPECTED_HEADER = [
'Name', 'Name',

View file

@ -1,8 +1,8 @@
import csv
import json import json
from collections import defaultdict from collections import defaultdict
from pprint import pp
import cube_json
def load_cube_from_txt(filename): def load_cube_from_txt(filename):
@ -37,48 +37,11 @@ EXPECTED_HEADER = [
] ]
COLUMNS = {column_name: i for i, column_name in enumerate(EXPECTED_HEADER)} COLUMNS = {column_name: i for i, column_name in enumerate(EXPECTED_HEADER)}
def load_cube_from_csv(filename, tag_filter=None):
cards = []
with open(filename) as f:
reader = csv.reader(f)
header_line = next(reader)
assert header_line == EXPECTED_HEADER
for line in reader:
card = line[COLUMNS['Name']]
# Export CSV is broken on CubeCobra because Image Back URL is always just one double quotes character.
tags = [t.strip() for t in line[COLUMNS['Tags']].split(', ')]
if tag_filter is None or any(t in tag_filter for t in tags):
cards.append(card)
return cards
def is_actual_mtg_card(card_dict): def is_actual_mtg_card(card_dict):
return card_dict['set_type'] not in {'memorabilia', 'funny', 'token'} return card_dict['set_type'] not in {'memorabilia', 'funny', 'token'}
def create_cube_json(cards):
# Call this once to create a smaller JSON file from the full card list
# Visit https://scryfall.com/docs/api/bulk-data and look for "Oracle Cards" file;
# Download and rename it to 'oracle_cards.json'
with open('oracle_cards.json', encoding='utf8') as f_oracle_cards:
d = json.load(f_oracle_cards)
output_filename = TEMP_JSON
f_cube_json = open(output_filename, 'w+', encoding='utf8')
cube_data = []
for card in d:
if card['name'] in cards:
for i in range(cards.count(card['name'])):
cube_data += [card]
elif 'card_faces' in card.keys() and is_actual_mtg_card(card):
for face in card['card_faces']:
if face['name'] in cards:
cube_data += [card]
string_data = json.dumps(cube_data)
f_cube_json.write(string_data)
f_cube_json.close()
return TEMP_JSON
def create_all_cards_json(): def create_all_cards_json():
with open('oracle_cards.json', encoding='utf8') as f_oracle_cards: with open('oracle_cards.json', encoding='utf8') as f_oracle_cards:
d = json.load(f_oracle_cards) d = json.load(f_oracle_cards)
@ -196,42 +159,46 @@ def get_text(card):
text += face['oracle_text'] + ' ' text += face['oracle_text'] + ' '
return text return text
################################################################################ if __name__ == '__main__':
################################################################################
full_oracle = False full_oracle = False
################################################################################ ################################################################################
# Use either line: # Use either line:
# 1. If you have a .txt of your cube # 1. If you have a .txt of your cube
# cube_list = load_cube_from_txt('YourCubeHere.txt') # cube_list = cube_json.load_cube_from_txt('YourCubeHere.txt')
# 2. If you have a .csv of your cube # 2. If you have a .csv of your cube
# cube_list = load_cube_from_csv('YourCubeHere.csv') # cube_list = cube_json.load_cube_from_csv('YourCubeHere.csv')
# 3. If you have a .csv of your cube and want to consider only cards with a certain tag # 3. If you have a .csv of your cube and want to consider only cards with a certain tag
cube_list = load_cube_from_csv('TheElegantCube_2020-11-17_5.0.4.csv', {'core'}) cube_list = cube_json.load_cube_from_csv('cube_csvs/TheElegantCube_2020-11-17_5.0.4.csv', {'core'})
################################################################################
# Calculate average of a cube ################################################################################
cube_json_handle = create_cube_json(cube_list) # Calculate average of a cube
cube_json_handle = cube_json.create_cube_json(cube_list)
# Summary of average words by color # Summary of average words by color
print_by_color(cube_json_handle, rank_cards=False, full_oracle=full_oracle) print_by_color(cube_json_handle, rank_cards=False, full_oracle=full_oracle)
# Individual cards by color, ranked by word count # Individual cards by color, ranked by word count
print_by_color(cube_json_handle, rank_cards=True, full_oracle=full_oracle) print_by_color(cube_json_handle, rank_cards=True, full_oracle=full_oracle)
# Average words excluding reminder text # Average words excluding reminder text
cube_word_count(cube_json_handle, full_oracle=False) cube_word_count(cube_json_handle, full_oracle=False)
# Average words in full oracle text # Average words in full oracle text
cube_word_count(cube_json_handle, full_oracle=True) cube_word_count(cube_json_handle, full_oracle=True)
################################################################################ ################################################################################
# Calculate average of all Magic cards
# Calculate average of all Magic cards # card_db_json_handle = create_all_cards_json()
# print_by_color(card_db_json_handle, rank_cards=True, full_oracle=full_oracle)
# cube_word_count(card_db_json_handle, full_oracle=full_oracle)
# card_db_json_handle = create_all_cards_json() ################################################################################
# print_by_color(card_db_json_handle, rank_cards=True, full_oracle=full_oracle)
# cube_word_count(card_db_json_handle, full_oracle=full_oracle) pass