-
Notifications
You must be signed in to change notification settings - Fork 0
Expand file tree
/
Copy pathprep_data.py
More file actions
95 lines (80 loc) · 3.54 KB
/
Copy pathprep_data.py
File metadata and controls
95 lines (80 loc) · 3.54 KB
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
# ============================================================
# BSAN 615 — Final Presentation: Data Pre-Processing
# Run this ONCE before final_charts.R
# Reads investments_VC.csv -> writes 6 clean helper CSVs
# ============================================================
import csv, statistics
from collections import defaultdict
rows = []
with open('investments_VC.csv', encoding='latin-1', newline='') as f:
reader = csv.DictReader(f)
for row in reader:
rows.append(row)
print(f"Loaded {len(rows)} rows from investments_VC.csv")
MKT_KEY = ' market '
FUND_KEY = ' funding_total_usd '
def clean_mkt(s): return s.strip() if s else ''
def parse_fund(s):
s = s.strip().replace(',','').replace(' ','').replace('"','')
if not s or set(s) <= {'-'}: return None
try:
v = float(s)
return v if v > 0 else None
except: return None
FOCUS_YEARS = [str(y) for y in range(2005, 2014)]
FOCUS_MKT = ['Software','Biotechnology','Mobile','E-Commerce',
'Clean Technology','Health Care','Finance',
'Social Media','Education','Games']
# Filter and enrich
vc = []
for r in rows:
yr = r.get('founded_year','').strip()
if yr not in FOCUS_YEARS: continue
vc.append({
'_market': clean_mkt(r[MKT_KEY]),
'_funding': parse_fund(r[FUND_KEY]),
'_year': int(yr),
'_seed': parse_fund(r.get('seed','')),
'_roundA': parse_fund(r.get('round_A','')),
'_status': r.get('status','').strip(),
})
print(f"Filtered to {len(vc)} rows (2005-2013)")
# Chart 1: company formations by year
with open('c1_formations.csv','w',newline='') as f:
w = csv.writer(f); w.writerow(['founded_year','companies'])
for yr in range(2005,2014):
w.writerow([yr, sum(1 for r in vc if r['_year']==yr)])
# Chart 2: all funding values (histogram)
with open('c2_hist.csv','w',newline='') as f:
w = csv.writer(f); w.writerow(['funding_usd'])
for r in vc:
if r['_funding']: w.writerow([r['_funding']])
# Chart 3: sector x year counts (heatmap)
with open('c3_heatmap.csv','w',newline='') as f:
w = csv.writer(f); w.writerow(['market','founded_year','n'])
for m in FOCUS_MKT:
for yr in range(2005,2014):
w.writerow([m, yr, sum(1 for r in vc if r['_market']==m and r['_year']==yr)])
# Chart 4: seed vs Series A funnel
with open('c4_funnel.csv','w',newline='') as f:
w = csv.writer(f); w.writerow(['founded_year','seed_cos','roundA_cos'])
for yr in range(2005,2014):
w.writerow([yr,
sum(1 for r in vc if r['_year']==yr and r['_seed']),
sum(1 for r in vc if r['_year']==yr and r['_roundA'])])
# Chart 5: median funding by sector (slope chart)
SLOPE_SECTORS = ['Software','Biotechnology','Mobile','E-Commerce',
'Clean Technology','Health Care','Finance','Social Media']
with open('c5_slope.csv','w',newline='') as f:
w = csv.writer(f); w.writerow(['market','founded_year','median_m'])
for m in SLOPE_SECTORS:
for yr in [2005,2013]:
vals = [r['_funding'] for r in vc if r['_market']==m and r['_year']==yr and r['_funding']]
if vals: w.writerow([m, yr, round(statistics.median(vals)/1e6, 4)])
# Chart 6: funding distribution (box & whisker)
with open('c6_box.csv','w',newline='') as f:
w = csv.writer(f); w.writerow(['market','funding_m'])
for r in vc:
if r['_market'] in SLOPE_SECTORS and r['_funding'] and r['_funding'] < 5e8:
w.writerow([r['_market'], round(r['_funding']/1e6, 4)])
print("All 6 helper CSVs written. Run final_charts.R next.")