Count the number of states in specific sections of 10-K

Viewed 25

I am trying to count the number of states in 10-K within specific sections. I was able to count the number of each states in whole 10-K text. But I would like to count the number of states only in specific sections. Specific section has unique name like Item 1. Business, Item 2. Properties, Item 6. Selected Financial, Item 7. Management

Is there anyway that I can add to my code to count the number states only in these sections? I want aggregated numbers count from these 4 sections, not separate counts.

import re, os, glob, csv
from bs4 import BeautifulSoup
from itertools import takewhile

path =
outputfile =

AL=re.compile(r'(\s[A][L].|\s[A][L],|Alabama)',re.I)
AK=re.compile(r'(\s[A][K].|\s[A][K],|Alaska)',re.I)
AZ=re.compile(r'(\s[A][Z]|Arizona)',re.I)
ITEM1=re.compile(r'(Item\s1.\sBusiness)s*',re.I)
ITEM2=re.compile(r'(Item\s2.\sProperties)s*',re.I)
ITEM3=re.compile(r'(Item\s3.\sLegal)s*',re.I)
ITEM6=re.compile(r'(Item\s6.\sSelected\sFinancial\sData)s*',re.I)
ITEM7=re.compile(r'(Item\s7.\sManagement\ss\sDiscussion\sand\sAnalysis)s*',re.I)
ITEM8=re.compile(r'(Item\s8.\sFinancial)s*',re.I)

t=0
data=list()
os.chdir(path)
\#Iterate through all the files in the 'path' directory
for i in glob.glob('*.*'):
    fd=i.split('.')[0].split('_')[0]
    cik=i.split('.')[0].split('_')[4]
    ft=i.split('.')[0].split('_')[1]
    doc=open(i,'r', encoding="utf-8").read()
    try:
        soup=BeautifulSoup(doc, "html.parser")
        \#Find all the text in the 10-K
        text=(''.join(soup.findAll(text=True)))
        num_AL = str(len(AL.findall(text)))
        num_AK = str(len(AK.findall(text)))
        num_AZ = str(len(AZ.findall(text)))
        data.append([cik,ft,fd,num_AL,num_AK,num_AZ])
    except:
        data.append([cik,ft,fd,'0','0','0','0'])
        continue

    #Output the data every 50 observations
    t+=1
    if t==50:
        fileobj = open(outputfile, 'a', newline='')
        csvfile = csv.writer(fileobj)
        for row in data:
            csvfile.writerow(row)
        fileobj.close()
        t=0
        data=list()
        continue

\#Write any remaining output to the csv file
fileobj = open(outputfile, 'a', newline='')
csvfile = csv.writer(fileobj)
for row in data:
    csvfile.writerow(row)
fileobj.close()
0 Answers
Related