99import os
1010import sys
1111import uuid
12+ from datetime import datetime
1213from io import BytesIO
1314from io import StringIO
1415
15- from spdx .checksum import Checksum
16- from spdx .checksum import ChecksumAlgorithm
17- from spdx .creationinfo import Tool
18- from spdx .document import ExtractedLicense
19- from spdx .document import Document
20- from spdx .license import License
21- from spdx .file import File
22- from spdx .package import Package
23- from spdx .relationship import Relationship
24- from spdx .utils import calc_verif_code
25- from spdx .utils import NoAssert
26- from spdx .utils import SPDXNone
27- from spdx .version import Version
2816
2917from license_expression import Licensing
3018from commoncode .cliutils import OUTPUT_GROUP
3119from commoncode .cliutils import PluggableCommandLineOption
3220from commoncode .fileutils import file_name
3321from commoncode .fileutils import parent_directory
3422from commoncode .text import python_safe_name
23+ from spdx_tools .spdx .model import SpdxNoAssertion , Version , CreationInfo , Actor , ActorType , Document , Package , File , \
24+ Checksum , ChecksumAlgorithm , ExtractedLicensingInfo , SpdxNone , Relationship , PackageVerificationCode , \
25+ RelationshipType
26+ from spdx_tools .spdx .spdx_element_utils import calculate_package_verification_code
27+
3528from formattedcode import FileOptionType
3629from licensedcode .detection import get_matches_from_detection_mappings
3730from plugincode .output import output_impl
@@ -63,50 +56,6 @@ def logger_debug(*args):
6356Output plugins to write scan results in SPDX format.
6457"""
6558
66- _spdx_list_is_patched = False
67-
68-
69- def _patch_license_list ():
70- """
71- Patch the SPDX Python library license list to match the list of ScanCode
72- known SPDX licenses.
73- """
74- global _spdx_list_is_patched
75- if not _spdx_list_is_patched :
76- from spdx .config import LICENSE_MAP
77- from licensedcode .models import load_licenses
78- licenses = load_licenses (with_deprecated = True )
79- spdx_licenses = get_licenses_by_spdx_key (licenses .values ())
80- LICENSE_MAP .update (spdx_licenses )
81- _spdx_list_is_patched = True
82-
83-
84- def get_licenses_by_spdx_key (licenses ):
85- """
86- Return a mapping of {spdx_key: license object} given a ``license`` sequence
87- of License objects.
88- """
89- spdx_licenses = {}
90- for lic in licenses :
91- if not (lic .spdx_license_key or lic .other_spdx_license_keys ):
92- continue
93-
94- if lic .spdx_license_key :
95- name = lic .name
96- slk = lic .spdx_license_key
97- spdx_licenses [slk ] = name
98- spdx_licenses [name ] = slk
99-
100- for other_spdx in lic .other_spdx_license_keys :
101- if not (other_spdx and other_spdx .strip ()):
102- continue
103- slk = other_spdx
104- spdx_licenses [slk ] = name
105- spdx_licenses [name ] = slk
106-
107- return spdx_licenses
108-
109-
11059@output_impl
11160class SpdxTvOutput (OutputPlugin ):
11261
@@ -220,7 +169,7 @@ def write_spdx(
220169 tool_version ,
221170 notice ,
222171 package_name = '' ,
223- download_location = NoAssert (),
172+ download_location = SpdxNoAssertion (),
224173 as_tagvalue = True ,
225174 spdx_version = (2 , 2 ),
226175 with_notice_text = False ,
@@ -240,37 +189,43 @@ def write_spdx(
240189 licensing = Licensing ()
241190
242191 as_rdf = not as_tagvalue
243- _patch_license_list ()
244192
245193 ns_prefix = '_' .join (package_name .lower ().split ())
246194 comment = notice + f'\n SPDX License List: { scancode_config .spdx_license_list_version } '
247195
248196 version_major , version_minor = scancode_config .spdx_license_list_version .split ("." )
249197 spdx_license_list_version = Version (major = version_major , minor = version_minor )
250198
251- doc = Document (
252- version = Version (* spdx_version ),
253- data_license = License .from_identifier ('CC0-1.0' ),
254- comment = notice ,
255- namespace = f'http://spdx.org/spdxdocs/{ ns_prefix } -{ uuid .uuid4 ()} ' ,
256- license_list_version = scancode_config .spdx_license_list_version ,
257- name = 'SPDX Document created by ScanCode Toolkit'
199+ tool_name = tool_name or 'ScanCode'
200+ creator = Actor (ActorType .TOOL , f'{ tool_name } { tool_version } ' )
201+
202+ creation_info = CreationInfo (
203+ spdx_id = "SPDXRef-DOCUMENT" ,
204+ spdx_version = f"SPDX-{ spdx_version [0 ]} .{ spdx_version [1 ]} " ,
205+ data_license = 'CC0-1.0' ,
206+ document_comment = comment ,
207+ document_namespace = f'http://spdx.org/spdxdocs/{ ns_prefix } -{ uuid .uuid4 ()} ' ,
208+ license_list_version = spdx_license_list_version ,
209+ name = 'SPDX Document created by ScanCode Toolkit' ,
210+ creators = [creator ],
211+ created = datetime .now (),
258212 )
259213
260- tool_name = tool_name or 'ScanCode'
261- doc .creation_info .add_creator (Tool (f'{ tool_name } { tool_version } ' ))
262- doc .creation_info .set_created_now ()
263- doc .creation_info .license_list_version = spdx_license_list_version
264214
265215 package_id = '001'
266- package = doc . package = Package (
216+ package = Package (
267217 name = package_name ,
268218 download_location = download_location ,
269219 spdx_id = f'SPDXRef-{ package_id } ' ,
270220 )
271221
222+ doc = Document (
223+ creation_info = creation_info ,
224+ packages = [package ],
225+ )
226+
272227 # Use a set of unique copyrights for the package.
273- package . cr_text = set ()
228+ package_copyright_texts = set ()
274229
275230 all_files_have_no_license = True
276231 all_files_have_no_copyright = True
@@ -285,13 +240,18 @@ def write_spdx(
285240 # Set a relative file name as that is what we want in
286241 # SPDX output (with explicit leading './').
287242 name = './' + file_data .get ('path' )
288- file_entry = File (
289- spdx_id = f'SPDXRef-{ sid } ' ,
290- name = name )
243+
291244 if file_data .get ('file_type' ) == 'empty' :
292- file_entry . set_checksum ( Checksum (ChecksumAlgorithm .SHA1 , "da39a3ee5e6b4b0d3255bfef95601890afd80709" ) )
245+ checksum = Checksum (ChecksumAlgorithm .SHA1 , "da39a3ee5e6b4b0d3255bfef95601890afd80709" )
293246 else :
294- file_entry .set_checksum (Checksum (ChecksumAlgorithm .SHA1 , file_data .get ('sha1' ) or '' ))
247+ # FIXME: this sets the checksum of a file to the empty string hash if unknown; tracked in https://github.com/nexB/scancode-toolkit/issues/3453
248+ checksum = Checksum (ChecksumAlgorithm .SHA1 , file_data .get ('sha1' ) or 'da39a3ee5e6b4b0d3255bfef95601890afd80709' )
249+
250+ file_entry = File (
251+ spdx_id = f'SPDXRef-{ sid } ' ,
252+ name = name ,
253+ checksums = [checksum ]
254+ )
295255
296256 file_license_detections = file_data .get ('license_detections' )
297257 license_matches = get_matches_from_detection_mappings (file_license_detections )
@@ -312,63 +272,63 @@ def write_spdx(
312272 spdx_id = f'LicenseRef-scancode-{ license_key } '
313273 is_license_ref = spdx_id .lower ().startswith ('licenseref-' )
314274
315- if not is_license_ref :
316- spdx_license = License .from_identifier (spdx_id )
317- else :
318- spdx_license = ExtractedLicense (spdx_id )
319- spdx_license .name = file_license .short_name
275+ spdx_license = licensing .parse (spdx_id )
276+
277+ if is_license_ref :
278+ text = match .get ('matched_text' )
320279 # FIXME: replace this with the licensedb URL
321280 comment = (
322281 f'See details at https://github.com/nexB/scancode-toolkit'
323282 f'/blob/develop/src/licensedcode/data/licenses/{ license_key } .yml\n '
324283 )
325- spdx_license .comment = comment
326- text = match .get ('matched_text' )
327- # always set some text, even if we did not extract the
328- # matched text
329- if not text :
330- text = comment
331- spdx_license .text = text
332- doc .add_extr_lic (spdx_license )
284+ extracted_license = ExtractedLicensingInfo (
285+ license_id = spdx_id ,
286+ # always set some text, even if we did not extract the
287+ # matched text
288+ extracted_text = text if text else comment ,
289+ license_name = file_license .short_name ,
290+ comment = comment ,
291+ )
292+ doc .extracted_licensing_info += [extracted_license ]
333293
334294 # Add licenses in the order they appear in the file. Maintaining
335295 # the order might be useful for provenance purposes.
336- file_entry .add_lics ( spdx_license )
337- package .add_lics_from_file ( spdx_license )
296+ file_entry .license_info_in_file += [ spdx_license ]
297+ package .license_info_from_files += [ spdx_license ]
338298
339299 elif license_matches is None :
340300 all_files_have_no_license = False
341- file_entry .add_lics ( NoAssert ())
301+ file_entry .license_info_in_file += [ SpdxNoAssertion ()]
342302
343303 else :
344- file_entry .add_lics ( SPDXNone ())
304+ file_entry .license_info_in_file += [ SpdxNone ()]
345305
346- file_entry .conc_lics = NoAssert ()
306+ file_entry .license_concluded = SpdxNoAssertion ()
347307
348308 file_copyrights = file_data .get ('copyrights' )
349309 if file_copyrights :
350310 all_files_have_no_copyright = False
351- file_entry . copyright = []
311+ copyrights = []
352312 for file_copyright in file_copyrights :
353- file_entry . copyright .append (file_copyright .get ('copyright' ))
313+ copyrights .append (file_copyright .get ('copyright' ))
354314
355- package . cr_text . update (file_entry . copyright )
315+ package_copyright_texts . update (copyrights )
356316
357317 # Create a text of copyright statements in the order they appear in
358318 # the file. Maintaining the order might be useful for provenance
359319 # purposes.
360- file_entry .copyright = '\n ' .join (file_entry . copyright ) + '\n '
320+ file_entry .copyright_text = '\n ' .join (copyrights ) + '\n '
361321
362322 elif file_copyrights is None :
363323 all_files_have_no_copyright = False
364- file_entry .copyright = NoAssert ()
324+ file_entry .copyright_text = SpdxNoAssertion ()
365325
366326 else :
367- file_entry .copyright = SPDXNone ()
327+ file_entry .copyright_text = SpdxNone ()
368328
369- doc .add_file ( file_entry )
370- relationship = Relationship (f' { package .spdx_id } CONTAINS { file_entry .spdx_id } ' )
371- doc .add_relationship ( relationship )
329+ doc .files += [ file_entry ]
330+ relationship = Relationship (package .spdx_id , RelationshipType . CONTAINS , file_entry .spdx_id )
331+ doc .relationships += [ relationship ]
372332
373333 if not doc .files :
374334 if as_tagvalue :
@@ -379,33 +339,29 @@ def write_spdx(
379339 output_file .write (msg )
380340
381341 # Remove duplicate licenses from the list for the package.
382- unique_licenses = {l .identifier : l for l in package .licenses_from_files }
383- unique_licenses = list (unique_licenses .values ())
384- if not len (package .licenses_from_files ):
342+ package .license_info_from_files = list (set (package .license_info_from_files ))
343+ if not package .license_info_from_files :
385344 if all_files_have_no_license :
386- package .licenses_from_files = [SPDXNone ()]
345+ package .license_info_from_files = [SpdxNone ()]
387346 else :
388- package .licenses_from_files = [NoAssert ()]
347+ package .license_info_from_files = [SpdxNoAssertion ()]
389348 else :
390349 # List license identifiers alphabetically for the package.
391- package .licenses_from_files = sorted (
392- unique_licenses ,
393- key = lambda x : x .identifier ,
394- )
350+ package .license_info_from_files = sorted (package .license_info_from_files )
395351
396- if len ( package . cr_text ) == 0 :
352+ if not package_copyright_texts :
397353 if all_files_have_no_copyright :
398- package .cr_text = SPDXNone ()
354+ package .copyright_text = SpdxNone ()
399355 else :
400- package .cr_text = NoAssert ()
356+ package .copyright_text = SpdxNoAssertion ()
401357 else :
402358 # Create a text of alphabetically sorted copyright
403359 # statements for the package.
404- package .cr_text = '\n ' .join (sorted (package . cr_text )) + '\n '
360+ package .copyright_text = '\n ' .join (sorted (package_copyright_texts )) + '\n '
405361
406- package .verif_code = calc_verif_code (doc .files )
407- package .license_declared = NoAssert ()
408- package .conc_lics = NoAssert ()
362+ package .verification_code = calculate_package_verification_code (doc .files )
363+ package .license_declared = SpdxNoAssertion ()
364+ package .license_concluded = SpdxNoAssertion ()
409365
410366 # The spdx-tools write_document returns either:
411367 # - unicode for tag values
@@ -416,19 +372,15 @@ def write_spdx(
416372 # in the other case we deal with text all the way.
417373
418374 if doc .files :
419-
420- if as_tagvalue :
421- from spdx .writers .tagvalue import write_document # NOQA
422- elif as_rdf :
423- from spdx .writers .rdf import write_document # NOQA
424-
425375 if as_tagvalue :
376+ from spdx_tools .spdx .writer .tagvalue .tagvalue_writer import write_document_to_stream # NOQA
426377 spdx_output = StringIO ()
427378 elif as_rdf :
379+ from spdx_tools .spdx .writer .rdf .rdf_writer import write_document_to_stream # NOQA
428380 # rdf is utf-encoded bytes
429381 spdx_output = BytesIO ()
430382
431- write_document (doc , spdx_output , validate = False )
383+ write_document_to_stream (doc , spdx_output , validate = False )
432384 result = spdx_output .getvalue ()
433385
434386 if as_rdf :
0 commit comments