Skip to content
Closed
Show file tree
Hide file tree
Changes from 13 commits
Commits
Show all changes
18 commits
Select commit Hold shift + click to select a range
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
203 changes: 178 additions & 25 deletions src/deltacode/__init__.py
Original file line number Diff line number Diff line change
Expand Up @@ -26,10 +26,12 @@
from __future__ import absolute_import

from collections import OrderedDict
import click

from deltacode.models import File
from deltacode.models import Scan
from deltacode import utils
from scancode.resource import VirtualCodebase


from pkg_resources import get_distribution, DistributionNotFound
Expand All @@ -48,14 +50,34 @@ class DeltaCode(object):
the form of File objects) contained in those scans.
"""
def __init__(self, new_path, old_path, options):
self.new = Scan(new_path)
self.old = Scan(old_path)
self.codebase1 = None
self.codebase2 = None

self.new_files_count = 0 #keeps the count of the new file
self.old_files_count = 0 #keeps the count of old files
self.new_files = [] # a list of [[new file1:Original path],[new file2:Original Path],...]
self.old_files = [] # a list of [[old file1:Original path],[old file2:Original Path],...]
self.new_files_fingerprint = dict() # map of { {new_file1:fingerprint},{new_file2:fingerprint},...} it will be needed when we need the fingerprints
self.old_files_fingerprint = dict() # map of { {old_file1:fingerprint},{old_file2:fingerprint},...}

Copy link
Copy Markdown
Contributor

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

What is all this stuff and why is it needed?

Copy link
Copy Markdown
Collaborator Author

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

self.new_files_count and self.old_files_count is to keep the track of the new files ,and old files.
I have added this to make the computation of the statistics easier,otherwise we would have to enumerate the virtual codebase objects to get the cont every time.
self.new_files_fingerprint and self.old_files_fingerprint it keeps a mapping of Resource objects files path from codebase1 and codebase2 with respect to the fingerprints which would be used in similarity comparisons.

Copy link
Copy Markdown
Collaborator Author

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

@MaJuRG self.new_files and self.old_files it is list of lists comprising of [Resource objects,with their original path] ,Now we need to keep the track of the original path in the align_scans for alignment of the files.

Copy link
Copy Markdown
Contributor

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

Its fine to enumerate on the Virtualcodebase. Stats should be calculated in the end anyway, and optionally for the user. I hate having carrying around all these unneeded fields.

Copy link
Copy Markdown
Contributor

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

Also, codebase objects have counts already that we can use. There is no need to track this twice.

Copy link
Copy Markdown
Collaborator Author

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

@MaJuRG , yes we can get rid of this self.new_files_count and self.old_files_count as they are already present in codebase objects as it is present in the headers of the codebase objects, but for the old files and new files and the fingerprint, I think it is better to have the enumeration done one time and cache those files in an array, else we will again need to enumerate it whenever required.

Copy link
Copy Markdown
Collaborator Author

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

@MaJuRG used the counts from the codebase , removed the additional variables for the files_count

self.new_files_original_path = dict() #this keeps a map of the path of file with respect to original path
self.old_files_original_path = dict()
self.options = options
self.deltas = []
self.errors = []
self.stats = Stat(self.new.files_count, self.old.files_count)

if self.new.path != '' and self.old.path != '':

try:
self.codebase1 = VirtualCodebase(new_path)
self.codebase2 = VirtualCodebase(old_path)
except Exception as exception:
self.errors.append(exception.message)
click.secho(exception.message ,fg = "red")

if self.codebase1 != None and self.codebase2 != None:

Copy link
Copy Markdown
Contributor

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

Can we just have a check for the inverse of this so we dont write everything under an unneeded indentation?

Copy link
Copy Markdown
Collaborator Author

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

@MaJuRG yeah , trying for it , I think it would be a better approach

self.fetch_files(self.codebase1,is_new = True)
self.fetch_files(self.codebase2,is_new = False)
self.stats = Stat(self.new_files_count, self.old_files_count)
self.new_files_errors = []
self.old_files_errors = []
self.determine_delta()
self.determine_moved()
self.license_diff()
Expand All @@ -67,6 +89,39 @@ def __init__(self, new_path, old_path, options):
self.deltas.sort(key=lambda Delta: Delta.factors, reverse=False)
self.deltas.sort(key=lambda Delta: Delta.score, reverse=True)

def fetch_files(self,codebase,is_new):
"""
Walk through the codebase, then generate the resources it(including all files and its directories)
then we enumerate over this generated codebase to get file, and directories as (obj)
Now during the time of enumeration we append files in the self.new_files list and incremants out self.new_files_count
Similarly for old_files.
Now , we also maintain a map which maps from object path to its fingerprint.
This map will be required when we calculate the hamming distances and compare similarity.
"""
resources = codebase.walk_filtered(topdown=True)
for i,obj in enumerate(resources):
if is_new:
# append in the new_files
self.new_files.append([obj,''])
try :
self.new_files_fingerprint[obj.path] = obj.fingerprint
except AttributeError:
self.new_files_fingerprint[obj.path] = None
if obj.is_file:
# increment the new files count
self.new_files_count += 1

Copy link
Copy Markdown
Contributor

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

Why not use dict.get() method instead of catching exception?

Copy link
Copy Markdown
Collaborator Author

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

@MaJuRG it would be nice , changing it

Copy link
Copy Markdown
Collaborator Author

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

@MaJuRG , the dict.get() method can not be used here as obj is not a dictionary it is a Virtualcodebase object.


else:
# append in the old files
self.old_files.append([obj,''])
try:
self.old_files_fingerprint[obj.path] = obj.fingerprint
except AttributeError:
self.old_files_fingerprint[obj.path] = None
if obj.is_file:
# increment the old files count
self.old_files_count += 1

def align_scans(self):
"""
Seek to align the paths of a pair of files (File objects) in the pair
Expand All @@ -75,12 +130,15 @@ def align_scans(self):
which calls utils.align_trees().
"""
try:
utils.fix_trees(self.new.files, self.old.files)
self.new_files_original_path , self.old_files_original_path = utils.fix_trees(self.new_files, self.old_files)
except utils.AlignmentException:
for f in self.new.files:
f.original_path = f.path
for f in self.old.files:
f.original_path = f.path
# self.new_files is a list of type [[ScannedResourceObject,originalPath],...]
# initially all original path are set to empty string
# so this part actually sets the original paths
for f in self.new_files:
f[1] = f[0].path
for f in self.old_files:
f[1] = f[0].path

def similarity(self):
"""
Expand All @@ -93,12 +151,14 @@ def similarity(self):
for delta in self.deltas:
if delta.new_file == None or delta.old_file == None:
continue
new_fingerprint = delta.new_file.fingerprint
old_fingerprint = delta.old_file.fingerprint
# this extracts the fingerprint corresponding to the particular file path

new_fingerprint = self.new_files_fingerprint.get(delta.new_file.path,None)
old_fingerprint = self.old_files_fingerprint.get(delta.old_file.path,None)
if new_fingerprint == None or old_fingerprint == None:
continue
new_fingerprint = utils.bitarray_from_hex(delta.new_file.fingerprint)
old_fingerprint = utils.bitarray_from_hex(delta.old_file.fingerprint)
new_fingerprint = utils.bitarray_from_hex(self.new_files_fingerprint[delta.new_file.path])
old_fingerprint = utils.bitarray_from_hex(self.old_files_fingerprint[delta.old_file.path])
hamming_distance = utils.hamming_distance(new_fingerprint, old_fingerprint)
if hamming_distance > 0 and hamming_distance <= SIMILARITY_LIMIT:
delta.score += hamming_distance
Expand All @@ -112,16 +172,15 @@ def determine_delta(self):
"""
# align scan and create our index
self.align_scans()
new_index = self.new.index_files()
old_index = self.old.index_files()

# returns file index wrt to old and new files
new_index = utils.index_files(self.new_files)
old_index = utils.index_files(self.old_files)
# gathering counts to ensure no files lost or missing from our 'deltas' set
new_visited, old_visited = 0, 0

# perform the deltas
for path, new_files in new_index.items():
for path, new_files in new_index.items():
for new_file in new_files:

if new_file.type != 'file':
continue

Expand Down Expand Up @@ -172,12 +231,12 @@ def determine_delta(self):
continue

# make sure everything is accounted for
if new_visited != self.new.files_count:
if new_visited != self.new_files_count:
self.errors.append(
'DeltaCode Warning: new_visited({}) != new_total({}). Assuming old scancode format.'.format(new_visited, self.new.files_count)
)

if old_visited != self.old.files_count:
if old_visited != self.old_files_count:
self.errors.append(
'DeltaCode Warning: old_visited({}) != old_total({}). Assuming old scancode format.'.format(old_visited, self.old.files_count)
)
Expand Down Expand Up @@ -237,6 +296,7 @@ def license_diff(self):
])

for delta in self.deltas:

utils.update_from_license_info(delta, unique_categories)

def copyright_diff(self):
Expand Down Expand Up @@ -323,7 +383,99 @@ def is_added(self):
if self.new_file and not self.old_file:
return True

def to_dict(self):
def copyrights_to_dict(self,file):
"""
Given a Copyright object, return an OrderedDict with the full
set of fields from the ScanCode 'copyrights' value.
"""
copyrightC = []
try :
copyrightC = file.copyrights
except AttributeError:
# arises when the ScannedResource do not have any license attribute
return []
if len(copyrightC) == 0:
return []
if isinstance(copyrightC[0],OrderedDict):
# all the copyright are in correct format
all_copyrights = []
for i in range(len(copyrightC)):
# we iterate over all the copyrights
statements = copyrightC[i].get("statements",None)
holders = copyrightC[i].get("holders",None)
d = OrderedDict([
('statements', statements),
('holders', holders)
])
all_copyrights.append(d)

return all_copyrights

def licenses_to_dict(self,file):
"""
Given a License object, return an OrderedDict with the full
set of fields from the ScanCode 'license' value.
"""
licenseL = []
try:
licenseL = file.licenses
except AttributeError:
# arises when the ScannedResource do not have any license attribute
return []

if len(licenseL) == 0:
return []
if isinstance(licenseL[0],OrderedDict):
# the licenses are in the correct format
all_licenses = []
for i in range(len(licenseL)):
# we iterate over all the licenses
key = licenseL[i].get("key",None)
score = licenseL[i].get("score",None)
short_key = licenseL[i].get("short_name",None)
category = licenseL[i].get("category",None)
owner = licenseL[i].get("owner",None)
d = OrderedDict([
('key', key),
('score', score),
('short_name', short_key),
('category', category),
('owner', owner)
])
all_licenses.append(d)
return all_licenses

def file_to_dict(self,deltacode, new_file = True):
if new_file==False and self.old_file :
return OrderedDict([
("path",self.old_file.path),
("type",self.old_file.type),
("name",self.old_file.name),
("size",self.old_file.size),
("sha1",self.old_file.sha1),
("fingerprint",deltacode.old_files_fingerprint.get(self.old_file.path,"")),
("original_path",deltacode.old_files_original_path.get(self.old_file.path, "")),
# since license itself has many sub fields so we obtain it from another utility function
("licenses",self.licenses_to_dict(self.old_file)),
# since copyright itself has many sub fields so we obtain it from another utility function
("copyrights",self.copyrights_to_dict(self.old_file))
])
elif new_file and self.new_file:
return OrderedDict([
("path",self.new_file.path),
("type",self.new_file.type),
("name",self.new_file.name),
("size",self.new_file.size),
("sha1",self.new_file.sha1),
("fingerprint",deltacode.new_files_fingerprint.get(self.new_file.path,"")),
("original_path",deltacode.new_files_original_path.get(self.new_file.path, "")),
# since license itself has many sub fields so we obtain it from another utility function
("licenses",self.licenses_to_dict(self.new_file)),
# since copyright itself has many sub fields so we obtain it from another utility function
("copyrights",self.copyrights_to_dict(self.new_file))
])

def to_dict(self,deltacode):
"""
Return an OrderedDict comprising the 'factors', 'score' and new and old
'path' attributes of the object.
Expand All @@ -342,8 +494,10 @@ def to_dict(self):
('status', self.status),
('factors', self.factors),
('score', self.score),
('new', new_file),
('old', old_file),
# receives the detail of the new file
('new', self.file_to_dict(deltacode , new_file = True)),
# receives the details of the old file
('old', self.file_to_dict(deltacode , new_file = False)),
])

class Stat(object):
Expand Down Expand Up @@ -389,4 +543,3 @@ def to_dict(self):
('percent_modified', self.percent_modified),
('percent_unmodified', self.percent_unmodified),
])

12 changes: 12 additions & 0 deletions src/deltacode/test_utils.py
Original file line number Diff line number Diff line change
Expand Up @@ -34,6 +34,7 @@
import json

from commoncode.system import on_windows
from scancode.resource import VirtualCodebase


def run_scan_click(options, monkeypatch=None, test_mode=True, expected_rc=0, env=None):
Expand Down Expand Up @@ -168,3 +169,14 @@ def streamline_headers(headers):
headers.pop('deltacode_version', None)
headers.pop('deltacode_options', None)
streamline_errors(headers['deltacode_errors'])


def fetch_files(location):
codebase = VirtualCodebase(location)

Copy link
Copy Markdown
Collaborator Author

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

This function has been extensively added to test_utils so as to prevent the creation of deltacode object for the test cases when only the Resource files are required

resourceFiles = []
resources = codebase.walk_filtered(topdown=True)

for index , obj in enumerate(resources):
resourceFiles.append([obj , ''])

return resourceFiles
Loading