-
Notifications
You must be signed in to change notification settings - Fork 0
Expand file tree
/
Copy pathLab6 Source Code.py
More file actions
112 lines (96 loc) · 2.98 KB
/
Copy pathLab6 Source Code.py
File metadata and controls
112 lines (96 loc) · 2.98 KB
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
import os
import math
import numpy as np
from nltk.stem.porter import *
# Read the folder
datasetname = {i: os.listdir("./dataset/" + i)for i in os.listdir("./dataset")}
#print(datasetname)
# Store all the words in a dictionary
data = dict()
for dirs in datasetname:
for i in datasetname.get(dirs):
if os.path.isfile(os.path.join("./dataset/", dirs, i)):
with open(os.path.join("./dataset/", dirs, i), 'r', encoding='Latin1') as fp:
data[os.path.join("./dataset/", dirs, i)] = re.split(r'(\W)+', fp.read())
#print(data)
# Read stopwords.txt and store in a set
stopwords = set()
with open("./stopwords.txt", 'r', encoding='utf-8') as gp:
stopwords = set(gp.read().split())
value = list()
for i in data.keys():
value.append(data[i])
newdata = list()
for i in value:
for w in i:
# Delete all non-alphabet characters and transform characters into lower case
wd = re.sub(r'[^a-z]', '', w.lower()).strip()
# Remove space and stopwords
if wd != '' and wd not in stopwords:
newdata.append(wd)
# Perform word stemming to remove the word suffix
stemmer = PorterStemmer()
plurals = newdata
singles = [stemmer.stem(plural) for plural in plurals]
#print(singles)
#print(len(singles))
# Define function fik
def fik(i, k):
document = dict()
#for dirs in datasetname:
#for i in datasetname.get(dirs):
with open(os.path.join(i), 'r', encoding="Latin1") as fp1:
document[os.path.join(i)] = re.split(r'(\W)+', fp1.read())
value1 = list()
for j in document.keys():
value1.append(document[j])
ndata = list()
for j in value1:
for x in j:
xd = re.sub(r'[^a-z]', '', x.lower()).strip()
if xd != '' and xd not in stopwords:
ndata.append(xd)
stemmer1 = PorterStemmer()
plurals1 = ndata
singles1 = [stemmer1.stem(plural) for plural in plurals1]
count = 0
for x in singles1:
if x == k:
count = count + 1
return count
# Calculate the number of documents
N = len(data)
#print(N)
# Define function nk
def nk(k):
count = 0
for i in data.keys():
if fik(i, k) != 0:
count = count + 1
return count
# Define function aik
def aik(i, k):
return fik(i, k)*math.log(N/nk(k), 10)
# Calculate the number of unique words
unique = set()
for x in singles:
unique.add(x)
D = len(unique)
#print(D)
# Define function Aik
def Aik(i, k):
sum = 0
for x in unique:
sum = sum + math.pow(aik(i, x), 2)
return aik(i, k)/math.pow(sum, 0.5)
# Store the document i in a list
address = list(data.keys())
#print(address)
#print(unique)
# Create the matrix
A = np.ones((N, D))
for i in address:
for k in unique:
A[i][k] = Aik(i, k)
# Save the matrix into npz document
np.savez('train-20ng.npz', X=A)