diff --git a/.gitattributes b/.gitattributes
new file mode 100644
index 0000000..c446d38
--- /dev/null
+++ b/.gitattributes
@@ -0,0 +1,2 @@
+# Ignore all Git auto CR/LF line endings conversions
+* binary
diff --git a/.gitignore b/.gitignore
index 0abbef1..68de2d2 100644
--- a/.gitignore
+++ b/.gitignore
@@ -67,3 +67,6 @@ pyvenv.cfg
/.pytest_cache/
lib64
tcl
+
+# Ignore Jupyter Notebook related temp files
+.ipynb_checkpoints/
diff --git a/.readthedocs.yml b/.readthedocs.yml
new file mode 100644
index 0000000..1b71cd9
--- /dev/null
+++ b/.readthedocs.yml
@@ -0,0 +1,18 @@
+# .readthedocs.yml
+# Read the Docs configuration file
+# See https://docs.readthedocs.io/en/stable/config-file/v2.html for details
+
+# Required
+version: 2
+
+# Where the Sphinx conf.py file is located
+sphinx:
+ configuration: docs/source/conf.py
+
+# Setting the python version and doc build requirements
+python:
+ install:
+ - method: pip
+ path: .
+ extra_requirements:
+ - docs
diff --git a/.travis.yml b/.travis.yml
index 23014f7..7a342df 100644
--- a/.travis.yml
+++ b/.travis.yml
@@ -1,6 +1,6 @@
# This is a skeleton Travis CI config file that provides a starting point for adding CI
# to a Python project. Since we primarily develop in python3, this skeleton config file
-# will be specific to that language.
+# will be specific to that language.
#
# See https://config.travis-ci.com/ for a full list of configuration options.
diff --git a/AUTHORS.rst b/AUTHORS.rst
index 51a19cc..00bd7dc 100644
--- a/AUTHORS.rst
+++ b/AUTHORS.rst
@@ -1,3 +1,11 @@
The following organizations or individuals have contributed to this repo:
--
+- Abhishek Kumar @Abhishek-Dev09
+- AlexB @a-tinsmith
+- Maximilian Huber @maxhbr
+- Michael Rupprecht @michaelrup
+- Philippe Ombredanne @pombredanne
+- Qingmin Duanmu @qduanmu
+- Rakesh Balusa @balusarakesh
+- Ravi Jain @JRavi2
+- Steven Esser @majurg
diff --git a/CHANGELOG.rst b/CHANGELOG.rst
index 5f8bc8d..d9183fc 100644
--- a/CHANGELOG.rst
+++ b/CHANGELOG.rst
@@ -1,5 +1,22 @@
Release notes
+=============
+
+vNext
+-----
+
+Version 21.1.15
+-------------
+
+*2021-01-15*
+- Drop support for Python 2
+- Use the latest CommonCode and TypeCode libraries
+
+*2020-11-13*
+- Add azure-pipelines CI support
+
+
+Version 20.10
-------------
-### Version 0.0.0
-*xxxx-xx-xx* -- Initial release.
+*2020-10-06*
+- Initial release.
diff --git a/MANIFEST.in b/MANIFEST.in
new file mode 100644
index 0000000..ef3721e
--- /dev/null
+++ b/MANIFEST.in
@@ -0,0 +1,15 @@
+graft src
+
+include *.LICENSE
+include NOTICE
+include *.ABOUT
+include *.toml
+include *.yml
+include *.rst
+include setup.*
+include configure*
+include requirements*
+include .git*
+
+global-exclude *.py[co] __pycache__ *.*~
+
diff --git a/NOTICE b/NOTICE
new file mode 100644
index 0000000..65936b2
--- /dev/null
+++ b/NOTICE
@@ -0,0 +1,19 @@
+#
+# Copyright (c) nexB Inc. and others.
+# SPDX-License-Identifier: Apache-2.0
+#
+# Visit https://aboutcode.org and https://github.com/nexB/ for support and download.
+# ScanCode is a trademark of nexB Inc.
+#
+# Licensed under the Apache License, Version 2.0 (the "License");
+# you may not use this file except in compliance with the License.
+# You may obtain a copy of the License at
+#
+# http://www.apache.org/licenses/LICENSE-2.0
+#
+# Unless required by applicable law or agreed to in writing, software
+# distributed under the License is distributed on an "AS IS" BASIS,
+# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+# See the License for the specific language governing permissions and
+# limitations under the License.
+#
diff --git a/README.rst b/README.rst
index 5065bc8..b0f2655 100644
--- a/README.rst
+++ b/README.rst
@@ -1,3 +1,38 @@
-extractcode
+ExtractCode
===========
-A set of functions and utilities used to extract archives.
+
+- license: Apache-2.0
+- copyright: copyright (c) nexB. Inc. and others
+- homepage_url: https://github.com/nexB/extractode
+- keywords: archiev, extraction, libarchive, 7zip, scancode-toolkit
+
+
+ExtractCode is a universal archive extractor. It uses behind the scenes
+the Python standard library, a custom ctypes binding to libarchive and
+the 7zip command line to extract a large number of common and
+less common archives and compressed files. It tries to extract things
+in the same way on all OSes, including auto-renaming files that would
+not have valid names on certain filesystems or when there are multiple
+copies of the same path in a given archive.
+The extraction is driven from a "voting" system that considers the
+file extension(s) and name, the file type and mime type (using a ctypes
+binding to libmagic) to select the most appropriate extractor or
+uncompressor function. It can handle multi-level archives such as tar.gz.
+
+
+
+Visit https://aboutcode.org and https://github.com/nexB/ for support and download.
+
+To set up the development environment::
+
+ source configure
+
+To run unit tests::
+
+ pytest -vvs -n 2
+
+To clean up development environment::
+
+ ./configure --clean
+
+
diff --git a/apache-2.0.LICENSE b/apache-2.0.LICENSE
index d9a10c0..261eeb9 100644
--- a/apache-2.0.LICENSE
+++ b/apache-2.0.LICENSE
@@ -174,3 +174,28 @@
of your accepting any such warranty or additional liability.
END OF TERMS AND CONDITIONS
+
+ APPENDIX: How to apply the Apache License to your work.
+
+ To apply the Apache License to your work, attach the following
+ boilerplate notice, with the fields enclosed by brackets "[]"
+ replaced with your own identifying information. (Don't include
+ the brackets!) The text should be enclosed in the appropriate
+ comment syntax for the file format. We also recommend that a
+ file or class name and description of purpose be included on the
+ same "printed page" as the copyright notice for easier
+ identification within third-party archives.
+
+ Copyright [yyyy] [name of copyright owner]
+
+ Licensed under the Apache License, Version 2.0 (the "License");
+ you may not use this file except in compliance with the License.
+ You may obtain a copy of the License at
+
+ http://www.apache.org/licenses/LICENSE-2.0
+
+ Unless required by applicable law or agreed to in writing, software
+ distributed under the License is distributed on an "AS IS" BASIS,
+ WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+ See the License for the specific language governing permissions and
+ limitations under the License.
diff --git a/azure-pipelines.yml b/azure-pipelines.yml
new file mode 100644
index 0000000..9a4c950
--- /dev/null
+++ b/azure-pipelines.yml
@@ -0,0 +1,64 @@
+
+################################################################################
+# We use Azure to run the full tests suites on multiple Python 3.x
+# on multiple Windows, macOS and Linux versions all on 64 bits
+# These jobs are using VMs with Azure-provided Python builds
+################################################################################
+
+jobs:
+
+ - template: etc/ci/azure-linux.yml
+ parameters:
+ job_name: ubuntu16_cpython
+ image_name: ubuntu-16.04
+ python_versions: ['3.6', '3.7', '3.8', '3.9']
+ test_suites:
+ all: tmp/bin/pytest -vvs
+
+ - template: etc/ci/azure-linux.yml
+ parameters:
+ job_name: ubuntu18_cpython
+ image_name: ubuntu-18.04
+ python_versions: ['3.6', '3.7', '3.8', '3.9']
+ test_suites:
+ all: tmp/bin/pytest -n 2 -vvs
+
+ - template: etc/ci/azure-linux.yml
+ parameters:
+ job_name: ubuntu20_cpython
+ image_name: ubuntu-20.04
+ python_versions: ['3.6', '3.7', '3.8', '3.9']
+ test_suites:
+ all: tmp/bin/pytest -n 2 -vvs
+
+ - template: etc/ci/azure-mac.yml
+ parameters:
+ job_name: macos1014_cpython
+ image_name: macos-10.14
+ python_versions: ['3.6', '3.7', '3.8', '3.9']
+ test_suites:
+ all: tmp/bin/pytest -n 2 -vvs
+
+ - template: etc/ci/azure-mac.yml
+ parameters:
+ job_name: macos1015_cpython
+ image_name: macos-10.15
+ python_versions: ['3.6', '3.7', '3.8', '3.9']
+ test_suites:
+ all: tmp/bin/pytest -n 2 -vvs
+
+ - template: etc/ci/azure-win.yml
+ parameters:
+ job_name: win2016_cpython
+ image_name: vs2017-win2016
+ python_versions: ['3.6', '3.7', '3.8', '3.9']
+ test_suites:
+ all: tmp\Scripts\pytest -n 2 -vvs
+
+ - template: etc/ci/azure-win.yml
+ parameters:
+ job_name: win2019_cpython
+ image_name: windows-2019
+ python_versions: ['3.6', '3.7', '3.8', '3.9']
+ test_suites:
+ all: tmp\Scripts\pytest -n 2 -vvs
diff --git a/configure b/configure
index a35c8c9..8f3a68e 100755
--- a/configure
+++ b/configure
@@ -1,6 +1,7 @@
#!/usr/bin/env bash
#
-# Copyright (c) nexB Inc. http://www.nexb.com/ - All rights reserved.
+# Copyright (c) nexB Inc. and others.
+# SPDX-License-Identifier: Apache-2.0
#
set -e
diff --git a/configure.bat b/configure.bat
new file mode 100644
index 0000000..f03ea07
--- /dev/null
+++ b/configure.bat
@@ -0,0 +1,121 @@
+@echo OFF
+@setlocal
+@rem Copyright (c) nexB Inc. http://www.nexb.com/ - All rights reserved.
+
+@rem ################################
+@rem # A configuration script for Windows
+@rem #
+@rem # The options and (optional) arguments are:
+@rem # --clean : this is exclusive of anything else and cleans the environment
+@rem # from built and installed files
+@rem #
+@rem # --python < path to python.exe> : this must be the first argument and set
+@rem # the path to the Python executable to use. If < path to python.exe> is
+@rem # set to "path", then the executable will be the python.exe available
+@rem # in the PATH.
+@rem ################################
+
+@rem Current directory where this .bat files lives
+set CFG_ROOT_DIR=%~dp0
+@rem path where a configured Python should live in the current virtualenv if installed
+set CONFIGURED_PYTHON=%CFG_ROOT_DIR%tmp\Scripts\python.exe
+set PYTHON_EXECUTABLE=
+
+
+@rem parse command line options and arguments
+:collectopts
+if "%1" EQU "--help" (goto cli_help)
+if "%1" EQU "--clean" (call rmdir /s /q "%CFG_ROOT_DIR%tmp") && call exit /b
+if "%1" EQU "--python" (set PROVIDED_PYTHON=%~2) && shift && shift && goto collectopts
+
+@rem If we have a pre-configured Python in our virtualenv, reuse this as-is and run
+if exist ""%CONFIGURED_PYTHON%"" (
+ set PYTHON_EXECUTABLE=%CONFIGURED_PYTHON%
+ goto run
+)
+
+@rem If we have a command arg for Python use this as-is
+if ""%PROVIDED_PYTHON%""==""path"" (
+ @rem use a bare python available in the PATH
+ set PYTHON_EXECUTABLE=python
+ goto run
+)
+if exist ""%PROVIDED_PYTHON%"" (
+ set PYTHON_EXECUTABLE=%PROVIDED_PYTHON%
+ goto run
+)
+
+
+@rem otherwise we search for a suitable Python interpreter
+:find_python
+@rem First check the existence of the "py" launcher (available in Python 3)
+@rem if we have it, check if we have a py -3 installed with the good version or a py 2.7
+@rem if not, check if we have an old py 2.7
+@rem exist if all fails
+
+where py >nul 2>nul
+if %ERRORLEVEL% == 0 (
+ @rem we have a py launcher, check for the availability of our required Python 3 version
+ py -3.6 --version >nul 2>nul
+ if %ERRORLEVEL% == 0 (
+ set PYTHON_EXECUTABLE=py -3.6
+ ) else (
+ @rem we have no required python 3, let's try python 2:
+ py -2 --version >nul 2>nul
+ if %ERRORLEVEL% == 0 (
+ set PYTHON_EXECUTABLE=py -2
+ ) else (
+ @rem we have py and no python 3 and 2, exit
+ echo * Unable to find an installation of Python.
+ exit /b 1
+ )
+ )
+) else (
+ @rem we have no py launcher, check for a default Python 2 installation
+ if not exist ""%DEFAULT_PYTHON2%"" (
+ echo * Unable to find an installation of Python.
+ exit /b 1
+ ) else (
+ set PYTHON_EXECUTABLE=%DEFAULT_PYTHON2%
+ )
+)
+
+
+:run
+@rem without this things may not always work on Windows 10, but this makes things slower
+set PYTHONDONTWRITEBYTECODE=1
+
+call mkdir "%CFG_ROOT_DIR%tmp"
+call curl -o "%CFG_ROOT_DIR%tmp\virtualenv.pyz" https://bootstrap.pypa.io/virtualenv.pyz
+call %PYTHON_EXECUTABLE% "%CFG_ROOT_DIR%tmp\virtualenv.pyz" "%CFG_ROOT_DIR%tmp"
+call "%CFG_ROOT_DIR%tmp\Scripts\activate"
+call "%CFG_ROOT_DIR%tmp\Scripts\pip" install --upgrade pip virtualenv setuptools wheel
+call "%CFG_ROOT_DIR%tmp\Scripts\pip" install -e .[testing]
+
+@rem Return a proper return code on failure
+if %ERRORLEVEL% neq 0 (
+ exit /b %ERRORLEVEL%
+)
+endlocal
+goto activate
+
+
+:cli_help
+echo A configuration script for Windows
+echo usage: configure [options] [path/to/config/directory]
+echo.
+echo The options and arguments are:
+echo --clean : this is exclusive of anything else and cleans the environment
+echo from built and installed files
+echo.
+echo --python path/to/python.exe : this is set to the path of an alternative
+echo Python executable to use. If path/to/python.exe is set to "path",
+echo then the executable will be the python.exe available in the PATH.
+echo.
+
+
+:activate
+@rem Activate the virtualenv
+if exist "%CFG_ROOT_DIR%tmp\Scripts\activate" (
+ "%CFG_ROOT_DIR%tmp\Scripts\activate"
+)
diff --git a/docs/Makefile b/docs/Makefile
new file mode 100644
index 0000000..d0c3cbf
--- /dev/null
+++ b/docs/Makefile
@@ -0,0 +1,20 @@
+# Minimal makefile for Sphinx documentation
+#
+
+# You can set these variables from the command line, and also
+# from the environment for the first two.
+SPHINXOPTS ?=
+SPHINXBUILD ?= sphinx-build
+SOURCEDIR = source
+BUILDDIR = build
+
+# Put it first so that "make" without argument is like "make help".
+help:
+ @$(SPHINXBUILD) -M help "$(SOURCEDIR)" "$(BUILDDIR)" $(SPHINXOPTS) $(O)
+
+.PHONY: help Makefile
+
+# Catch-all target: route all unknown targets to Sphinx using the new
+# "make mode" option. $(O) is meant as a shortcut for $(SPHINXOPTS).
+%: Makefile
+ @$(SPHINXBUILD) -M $@ "$(SOURCEDIR)" "$(BUILDDIR)" $(SPHINXOPTS) $(O)
diff --git a/docs/make.bat b/docs/make.bat
new file mode 100644
index 0000000..6247f7e
--- /dev/null
+++ b/docs/make.bat
@@ -0,0 +1,35 @@
+@ECHO OFF
+
+pushd %~dp0
+
+REM Command file for Sphinx documentation
+
+if "%SPHINXBUILD%" == "" (
+ set SPHINXBUILD=sphinx-build
+)
+set SOURCEDIR=source
+set BUILDDIR=build
+
+if "%1" == "" goto help
+
+%SPHINXBUILD% >NUL 2>NUL
+if errorlevel 9009 (
+ echo.
+ echo.The 'sphinx-build' command was not found. Make sure you have Sphinx
+ echo.installed, then set the SPHINXBUILD environment variable to point
+ echo.to the full path of the 'sphinx-build' executable. Alternatively you
+ echo.may add the Sphinx directory to PATH.
+ echo.
+ echo.If you don't have Sphinx installed, grab it from
+ echo.http://sphinx-doc.org/
+ exit /b 1
+)
+
+%SPHINXBUILD% -M %1 %SOURCEDIR% %BUILDDIR% %SPHINXOPTS% %O%
+goto end
+
+:help
+%SPHINXBUILD% -M help %SOURCEDIR% %BUILDDIR% %SPHINXOPTS% %O%
+
+:end
+popd
diff --git a/docs/scripts/doc8_style_check.sh b/docs/scripts/doc8_style_check.sh
new file mode 100644
index 0000000..9416323
--- /dev/null
+++ b/docs/scripts/doc8_style_check.sh
@@ -0,0 +1,5 @@
+#!/bin/bash
+# halt script on error
+set -e
+# Check for Style Code Violations
+doc8 --max-line-length 100 source --ignore D000 --quiet
\ No newline at end of file
diff --git a/docs/scripts/sphinx_build_link_check.sh b/docs/scripts/sphinx_build_link_check.sh
new file mode 100644
index 0000000..c542686
--- /dev/null
+++ b/docs/scripts/sphinx_build_link_check.sh
@@ -0,0 +1,5 @@
+#!/bin/bash
+# halt script on error
+set -e
+# Build locally, and then check links
+sphinx-build -E -W -b linkcheck source build
\ No newline at end of file
diff --git a/docs/source/conf.py b/docs/source/conf.py
new file mode 100644
index 0000000..529cae3
--- /dev/null
+++ b/docs/source/conf.py
@@ -0,0 +1,63 @@
+# Configuration file for the Sphinx documentation builder.
+#
+# This file only contains a selection of the most common options. For a full
+# list see the documentation:
+# https://www.sphinx-doc.org/en/master/usage/configuration.html
+
+# -- Path setup --------------------------------------------------------------
+
+# If extensions (or modules to document with autodoc) are in another directory,
+# add these directories to sys.path here. If the directory is relative to the
+# documentation root, use os.path.abspath to make it absolute, like shown here.
+#
+# import os
+# import sys
+# sys.path.insert(0, os.path.abspath('.'))
+
+
+# -- Project information -----------------------------------------------------
+
+project = 'nexb-skeleton'
+copyright = 'nexb Inc.'
+author = 'nexb Inc.'
+
+
+# -- General configuration ---------------------------------------------------
+
+# Add any Sphinx extension module names here, as strings. They can be
+# extensions coming with Sphinx (named 'sphinx.ext.*') or your custom
+# ones.
+extensions = [
+]
+
+# Add any paths that contain templates here, relative to this directory.
+templates_path = ['_templates']
+
+# List of patterns, relative to source directory, that match files and
+# directories to ignore when looking for source files.
+# This pattern also affects html_static_path and html_extra_path.
+exclude_patterns = []
+
+
+# -- Options for HTML output -------------------------------------------------
+
+# The theme to use for HTML and HTML Help pages. See the documentation for
+# a list of builtin themes.
+#
+html_theme = 'sphinx_rtd_theme'
+
+# Add any paths that contain custom static files (such as style sheets) here,
+# relative to this directory. They are copied after the builtin static files,
+# so a file named "default.css" will overwrite the builtin "default.css".
+html_static_path = ['_static']
+
+html_context = {
+ 'css_files': [
+ '_static/theme_overrides.css', # override wide tables in RTD theme
+ ],
+ "display_github": True,
+ "github_user": "nexB",
+ "github_repo": "nexb-skeleton",
+ "github_version": "develop", # branch
+ "conf_py_path": "/docs/source/", # path in the checkout to the docs root
+ }
\ No newline at end of file
diff --git a/docs/source/index.rst b/docs/source/index.rst
new file mode 100644
index 0000000..67fcf21
--- /dev/null
+++ b/docs/source/index.rst
@@ -0,0 +1,15 @@
+Welcome to nexb-skeleton's documentation!
+=========================================
+
+.. toctree::
+ :maxdepth: 2
+ :caption: Contents:
+
+ skeleton/index
+
+Indices and tables
+==================
+
+* :ref:`genindex`
+* :ref:`modindex`
+* :ref:`search`
diff --git a/docs/source/skeleton/index.rst b/docs/source/skeleton/index.rst
new file mode 100644
index 0000000..7dfc6cb
--- /dev/null
+++ b/docs/source/skeleton/index.rst
@@ -0,0 +1,15 @@
+# Docs Structure Guide
+# Rst docs - https://docutils.sourceforge.io/docs/ref/rst/restructuredtext.html
+#
+# 1. Place docs in folders under source for different sections
+# 2. Link them by adding individual index files in each section
+# to the main index, and then files for each section to their
+# respective index files.
+# 3. Use `.. include` statements to include other .rst files
+# or part of them, or use hyperlinks to a section of the docs,
+# to get rid of repetition.
+# https://docutils.sourceforge.io/docs/ref/rst/directives.html#including-an-external-document-fragment
+#
+# Note: Replace these guide/placeholder docs
+
+.. include:: ../../../README.rst
diff --git a/etc/ci/azure-linux.yml b/etc/ci/azure-linux.yml
new file mode 100644
index 0000000..752ae2e
--- /dev/null
+++ b/etc/ci/azure-linux.yml
@@ -0,0 +1,36 @@
+parameters:
+ job_name: ''
+ image_name: ''
+ python_versions: []
+ test_suites: {}
+ python_architecture: x64
+
+jobs:
+ - job: ${{ parameters.job_name }}
+
+ pool:
+ vmImage: ${{ parameters.image_name }}
+
+ strategy:
+ matrix:
+ ${{ each pyver in parameters.python_versions }}:
+ ${{ each tsuite in parameters.test_suites }}:
+ ${{ format('py{0} {1}', pyver, tsuite.key) }}:
+ python_version: ${{ pyver }}
+ test_suite_label: ${{ tsuite.key }}
+ test_suite: ${{ tsuite.value }}
+ steps:
+ - checkout: self
+ fetchDepth: 10
+
+ - task: UsePythonVersion@0
+ inputs:
+ versionSpec: '$(python_version)'
+ architecture: '${{ parameters.python_architecture }}'
+ displayName: 'Install Python $(python_version)'
+
+ - script: ./configure
+ displayName: 'Run Configure'
+
+ - script: $(test_suite)
+ displayName: 'Run $(test_suite_label) tests with py$(python_version) on ${{ parameters.job_name }}'
diff --git a/etc/ci/azure-mac.yml b/etc/ci/azure-mac.yml
new file mode 100644
index 0000000..752ae2e
--- /dev/null
+++ b/etc/ci/azure-mac.yml
@@ -0,0 +1,36 @@
+parameters:
+ job_name: ''
+ image_name: ''
+ python_versions: []
+ test_suites: {}
+ python_architecture: x64
+
+jobs:
+ - job: ${{ parameters.job_name }}
+
+ pool:
+ vmImage: ${{ parameters.image_name }}
+
+ strategy:
+ matrix:
+ ${{ each pyver in parameters.python_versions }}:
+ ${{ each tsuite in parameters.test_suites }}:
+ ${{ format('py{0} {1}', pyver, tsuite.key) }}:
+ python_version: ${{ pyver }}
+ test_suite_label: ${{ tsuite.key }}
+ test_suite: ${{ tsuite.value }}
+ steps:
+ - checkout: self
+ fetchDepth: 10
+
+ - task: UsePythonVersion@0
+ inputs:
+ versionSpec: '$(python_version)'
+ architecture: '${{ parameters.python_architecture }}'
+ displayName: 'Install Python $(python_version)'
+
+ - script: ./configure
+ displayName: 'Run Configure'
+
+ - script: $(test_suite)
+ displayName: 'Run $(test_suite_label) tests with py$(python_version) on ${{ parameters.job_name }}'
diff --git a/etc/ci/azure-win.yml b/etc/ci/azure-win.yml
new file mode 100644
index 0000000..6220857
--- /dev/null
+++ b/etc/ci/azure-win.yml
@@ -0,0 +1,36 @@
+parameters:
+ job_name: ''
+ image_name: ''
+ python_versions: []
+ test_suites: {}
+ python_architecture: x86
+
+jobs:
+ - job: ${{ parameters.job_name }}
+
+ pool:
+ vmImage: ${{ parameters.image_name }}
+
+ strategy:
+ matrix:
+ ${{ each pyver in parameters.python_versions }}:
+ ${{ each tsuite in parameters.test_suites }}:
+ ${{ format('py{0} {1}', pyver, tsuite.key) }}:
+ python_version: ${{ pyver }}
+ test_suite_label: ${{ tsuite.key }}
+ test_suite: ${{ tsuite.value }}
+ steps:
+ - checkout: self
+ fetchDepth: 10
+
+ - task: UsePythonVersion@0
+ inputs:
+ versionSpec: '$(python_version)'
+ architecture: '${{ parameters.python_architecture }}'
+ displayName: 'Install Python $(python_version)'
+
+ - script: configure --python path
+ displayName: 'Run Configure'
+
+ - script: $(test_suite)
+ displayName: 'Run $(test_suite_label) tests with py$(python_version) on ${{ parameters.job_name }}'
diff --git a/pyproject.toml b/pyproject.toml
index e75f1ce..55fb92c 100644
--- a/pyproject.toml
+++ b/pyproject.toml
@@ -36,11 +36,11 @@ norecursedirs = [
python_files = "*.py"
-python_classes="Test"
-python_functions="test"
+python_classes = "Test"
+python_functions = "test"
addopts = [
"-rfExXw",
"--strict",
"--doctest-modules"
-]
\ No newline at end of file
+]
diff --git a/setup.cfg b/setup.cfg
index 98a95ff..b2b4cae 100644
--- a/setup.cfg
+++ b/setup.cfg
@@ -1,21 +1,32 @@
-[wheel]
-universal=1
-
[metadata]
-license_file = apache-2.0.LICENSE
+license_files =
+ apache-2.0.LICENSE
+ NOTICE
+ AUTHORS.rst
+ CHANGELOG.rst
name = extractcode
author = nexB. Inc. and others
author_email = info@aboutcode.org
+license = Apache-2.0
+
+# description must be on ONE line https://github.com/pypa/setuptools/issues/1390
description = A set of utilities and functions for extracting archives.
long_description = file:README.rst
url = https://github.com/nexB/extractcode
classifiers =
Development Status :: 5 - Production/Stable
Intended Audience :: Developers
- Programming Language :: Python :: 2
Programming Language :: Python :: 3
+ Programming Language :: Python :: 3 :: Only
Topic :: Software Development
Topic :: Utilities
+keywords =
+ utilities
+ archive
+ extraction
+ libarchive
+ 7zip
+ scancode-toolkit
[options]
package_dir=
@@ -24,9 +35,9 @@ packages=find:
include_package_data = true
zip_safe = false
install_requires =
- commoncode >= 20.10
+ commoncode >= 21.1
plugincode >= 20.09
- typecode >= 20.10
+ typecode >= 21.1
typecode-libmagic >= 5.39.1.1
patch >= 1.15
extractcode-7z >= 16.5
@@ -42,6 +53,10 @@ testing =
# upstream
pytest >= 6
pytest-xdist >= 2
+docs=
+ Sphinx>=3.3.1
+ sphinx-rtd-theme>=0.5.0
+ doc8>=0.8.1
[options.entry_points]
console-scripts =
diff --git a/setup.py b/setup.py
index 45f160d..bac24a4 100644
--- a/setup.py
+++ b/setup.py
@@ -3,4 +3,4 @@
import setuptools
if __name__ == "__main__":
- setuptools.setup()
\ No newline at end of file
+ setuptools.setup()
diff --git a/src/README.rst b/src/README.rst
new file mode 100644
index 0000000..ec651fc
--- /dev/null
+++ b/src/README.rst
@@ -0,0 +1,2 @@
+Put your Python source code (and installable data) in this directory.
+
diff --git a/src/extractcode/NOTICE b/src/extractcode/NOTICE
index e7eb6d3..65936b2 100644
--- a/src/extractcode/NOTICE
+++ b/src/extractcode/NOTICE
@@ -1,56 +1,19 @@
-Software license
-================
-
-Copyright (c) 2017 nexB Inc. and others. All rights reserved.
-http://nexb.com and https://github.com/nexB/scancode-toolkit/
-The ScanCode software is licensed under the Apache License version 2.0.
-Data generated with ScanCode require an acknowledgment.
-ScanCode is a trademark of nexB Inc.
-
-You may not use this software except in compliance with the License.
-You may obtain a copy of the License at: http://apache.org/licenses/LICENSE-2.0
-Unless required by applicable law or agreed to in writing, software distributed
-under the License is distributed on an "AS IS" BASIS, WITHOUT WARRANTIES OR
-CONDITIONS OF ANY KIND, either express or implied. See the License for the
-specific language governing permissions and limitations under the License.
-
-When you publish or redistribute any data created with ScanCode or any ScanCode
-derivative work, you must accompany this data with the following acknowledgment:
-
- Generated with ScanCode and provided on an "AS IS" BASIS, WITHOUT WARRANTIES
- OR CONDITIONS OF ANY KIND, either express or implied. No content created from
- ScanCode should be considered or used as legal advice. Consult an Attorney
- for any legal advice.
- ScanCode is a free software code scanning tool from nexB Inc. and others.
- Visit https://github.com/nexB/scancode-toolkit/ for support and download.
-
-
-Third-party software licenses
-=============================
-
-ScanCode embeds third-party free and open source software packages under various
-licenses including copyleft licenses. Some of the third-party software packages
-are delivered as pre-built binaries. The origin and license of these packages is
-documented by .ABOUT files.
-
-The corresponding source code for pre-compiled third-party software is available
-for immediate download from the same release page where you obtained ScanCode at:
-https://github.com/nexB/scancode-toolkit/
-or https://github.com/nexB/scancode-thirdparty-src/
-
-You may also contact us to request the source code by email at info@nexb.com or
-by postal mail at:
-
- nexB Inc., ScanCode open source code request
- 735 Industrial Road, Suite #101, 94070 San Carlos, CA, USA
-
-Please indicate in your communication the ScanCode version for which you are
-requesting source code.
-
-
-License for ScanCode datasets
-=============================
-
-ScanCode includes datasets (e.g. for license detection) that are dedicated
-to the Public Domain using the Creative Commons CC0 1.0 Universal (CC0 1.0)
-Public Domain Dedication: http://creativecommons.org/publicdomain/zero/1.0/
+#
+# Copyright (c) nexB Inc. and others.
+# SPDX-License-Identifier: Apache-2.0
+#
+# Visit https://aboutcode.org and https://github.com/nexB/ for support and download.
+# ScanCode is a trademark of nexB Inc.
+#
+# Licensed under the Apache License, Version 2.0 (the "License");
+# you may not use this file except in compliance with the License.
+# You may obtain a copy of the License at
+#
+# http://www.apache.org/licenses/LICENSE-2.0
+#
+# Unless required by applicable law or agreed to in writing, software
+# distributed under the License is distributed on an "AS IS" BASIS,
+# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+# See the License for the specific language governing permissions and
+# limitations under the License.
+#
diff --git a/src/extractcode/PSF.LICENSE b/src/extractcode/PSF.LICENSE
deleted file mode 100644
index f472924..0000000
--- a/src/extractcode/PSF.LICENSE
+++ /dev/null
@@ -1,635 +0,0 @@
-A. HISTORY OF THE SOFTWARE
-==========================
-
-Python was created in the early 1990s by Guido van Rossum at Stichting
-Mathematisch Centrum (CWI, see http://www.cwi.nl) in the Netherlands
-as a successor of a language called ABC. Guido remains Python's
-principal author, although it includes many contributions from others.
-
-In 1995, Guido continued his work on Python at the Corporation for
-National Research Initiatives (CNRI, see http://www.cnri.reston.va.us)
-in Reston, Virginia where he released several versions of the
-software.
-
-In May 2000, Guido and the Python core development team moved to
-BeOpen.com to form the BeOpen PythonLabs team. In October of the same
-year, the PythonLabs team moved to Digital Creations (now Zope
-Corporation, see http://www.zope.com). In 2001, the Python Software
-Foundation (PSF, see http://www.python.org/psf/) was formed, a
-non-profit organization created specifically to own Python-related
-Intellectual Property. Zope Corporation is a sponsoring member of
-the PSF.
-
-All Python releases are Open Source (see http://www.opensource.org for
-the Open Source Definition). Historically, most, but not all, Python
-releases have also been GPL-compatible; the table below summarizes
-the various releases.
-
- Release Derived Year Owner GPL-
- from compatible? (1)
-
- 0.9.0 thru 1.2 1991-1995 CWI yes
- 1.3 thru 1.5.2 1.2 1995-1999 CNRI yes
- 1.6 1.5.2 2000 CNRI no
- 2.0 1.6 2000 BeOpen.com no
- 1.6.1 1.6 2001 CNRI yes (2)
- 2.1 2.0+1.6.1 2001 PSF no
- 2.0.1 2.0+1.6.1 2001 PSF yes
- 2.1.1 2.1+2.0.1 2001 PSF yes
- 2.2 2.1.1 2001 PSF yes
- 2.1.2 2.1.1 2002 PSF yes
- 2.1.3 2.1.2 2002 PSF yes
- 2.2.1 2.2 2002 PSF yes
- 2.2.2 2.2.1 2002 PSF yes
- 2.2.3 2.2.2 2003 PSF yes
- 2.3 2.2.2 2002-2003 PSF yes
- 2.3.1 2.3 2002-2003 PSF yes
- 2.3.2 2.3.1 2002-2003 PSF yes
- 2.3.3 2.3.2 2002-2003 PSF yes
- 2.3.4 2.3.3 2004 PSF yes
- 2.3.5 2.3.4 2005 PSF yes
- 2.4 2.3 2004 PSF yes
- 2.4.1 2.4 2005 PSF yes
- 2.4.2 2.4.1 2005 PSF yes
- 2.4.3 2.4.2 2006 PSF yes
- 2.4.4 2.4.3 2006 PSF yes
- 2.5 2.4 2006 PSF yes
- 2.5.1 2.5 2007 PSF yes
- 2.5.2 2.5.2 2008 PSF yes
-
-Footnotes:
-
-(1) GPL-compatible doesn't mean that we're distributing Python under
- the GPL. All Python licenses, unlike the GPL, let you distribute
- a modified version without making your changes open source. The
- GPL-compatible licenses make it possible to combine Python with
- other software that is released under the GPL; the others don't.
-
-(2) According to Richard Stallman, 1.6.1 is not GPL-compatible,
- because its license has a choice of law clause. According to
- CNRI, however, Stallman's lawyer has told CNRI's lawyer that 1.6.1
- is "not incompatible" with the GPL.
-
-Thanks to the many outside volunteers who have worked under Guido's
-direction to make these releases possible.
-
-
-B. TERMS AND CONDITIONS FOR ACCESSING OR OTHERWISE USING PYTHON
-===============================================================
-
-PYTHON SOFTWARE FOUNDATION LICENSE VERSION 2
---------------------------------------------
-
-1. This LICENSE AGREEMENT is between the Python Software Foundation
-("PSF"), and the Individual or Organization ("Licensee") accessing and
-otherwise using this software ("Python") in source or binary form and
-its associated documentation.
-
-2. Subject to the terms and conditions of this License Agreement, PSF
-hereby grants Licensee a nonexclusive, royalty-free, world-wide
-license to reproduce, analyze, test, perform and/or display publicly,
-prepare derivative works, distribute, and otherwise use Python
-alone or in any derivative version, provided, however, that PSF's
-License Agreement and PSF's notice of copyright, i.e., "Copyright (c)
-2001, 2002, 2003, 2004, 2005, 2006, 2007, 2008 Python Software Foundation;
-All Rights Reserved" are retained in Python alone or in any derivative
-version prepared by Licensee.
-
-3. In the event Licensee prepares a derivative work that is based on
-or incorporates Python or any part thereof, and wants to make
-the derivative work available to others as provided herein, then
-Licensee hereby agrees to include in any such work a brief summary of
-the changes made to Python.
-
-4. PSF is making Python available to Licensee on an "AS IS"
-basis. PSF MAKES NO REPRESENTATIONS OR WARRANTIES, EXPRESS OR
-IMPLIED. BY WAY OF EXAMPLE, BUT NOT LIMITATION, PSF MAKES NO AND
-DISCLAIMS ANY REPRESENTATION OR WARRANTY OF MERCHANTABILITY OR FITNESS
-FOR ANY PARTICULAR PURPOSE OR THAT THE USE OF PYTHON WILL NOT
-INFRINGE ANY THIRD PARTY RIGHTS.
-
-5. PSF SHALL NOT BE LIABLE TO LICENSEE OR ANY OTHER USERS OF PYTHON
-FOR ANY INCIDENTAL, SPECIAL, OR CONSEQUENTIAL DAMAGES OR LOSS AS
-A RESULT OF MODIFYING, DISTRIBUTING, OR OTHERWISE USING PYTHON,
-OR ANY DERIVATIVE THEREOF, EVEN IF ADVISED OF THE POSSIBILITY THEREOF.
-
-6. This License Agreement will automatically terminate upon a material
-breach of its terms and conditions.
-
-7. Nothing in this License Agreement shall be deemed to create any
-relationship of agency, partnership, or joint venture between PSF and
-Licensee. This License Agreement does not grant permission to use PSF
-trademarks or trade name in a trademark sense to endorse or promote
-products or services of Licensee, or any third party.
-
-8. By copying, installing or otherwise using Python, Licensee
-agrees to be bound by the terms and conditions of this License
-Agreement.
-
-
-BEOPEN.COM LICENSE AGREEMENT FOR PYTHON 2.0
--------------------------------------------
-
-BEOPEN PYTHON OPEN SOURCE LICENSE AGREEMENT VERSION 1
-
-1. This LICENSE AGREEMENT is between BeOpen.com ("BeOpen"), having an
-office at 160 Saratoga Avenue, Santa Clara, CA 95051, and the
-Individual or Organization ("Licensee") accessing and otherwise using
-this software in source or binary form and its associated
-documentation ("the Software").
-
-2. Subject to the terms and conditions of this BeOpen Python License
-Agreement, BeOpen hereby grants Licensee a non-exclusive,
-royalty-free, world-wide license to reproduce, analyze, test, perform
-and/or display publicly, prepare derivative works, distribute, and
-otherwise use the Software alone or in any derivative version,
-provided, however, that the BeOpen Python License is retained in the
-Software, alone or in any derivative version prepared by Licensee.
-
-3. BeOpen is making the Software available to Licensee on an "AS IS"
-basis. BEOPEN MAKES NO REPRESENTATIONS OR WARRANTIES, EXPRESS OR
-IMPLIED. BY WAY OF EXAMPLE, BUT NOT LIMITATION, BEOPEN MAKES NO AND
-DISCLAIMS ANY REPRESENTATION OR WARRANTY OF MERCHANTABILITY OR FITNESS
-FOR ANY PARTICULAR PURPOSE OR THAT THE USE OF THE SOFTWARE WILL NOT
-INFRINGE ANY THIRD PARTY RIGHTS.
-
-4. BEOPEN SHALL NOT BE LIABLE TO LICENSEE OR ANY OTHER USERS OF THE
-SOFTWARE FOR ANY INCIDENTAL, SPECIAL, OR CONSEQUENTIAL DAMAGES OR LOSS
-AS A RESULT OF USING, MODIFYING OR DISTRIBUTING THE SOFTWARE, OR ANY
-DERIVATIVE THEREOF, EVEN IF ADVISED OF THE POSSIBILITY THEREOF.
-
-5. This License Agreement will automatically terminate upon a material
-breach of its terms and conditions.
-
-6. This License Agreement shall be governed by and interpreted in all
-respects by the law of the State of California, excluding conflict of
-law provisions. Nothing in this License Agreement shall be deemed to
-create any relationship of agency, partnership, or joint venture
-between BeOpen and Licensee. This License Agreement does not grant
-permission to use BeOpen trademarks or trade names in a trademark
-sense to endorse or promote products or services of Licensee, or any
-third party. As an exception, the "BeOpen Python" logos available at
-http://www.pythonlabs.com/logos.html may be used according to the
-permissions granted on that web page.
-
-7. By copying, installing or otherwise using the software, Licensee
-agrees to be bound by the terms and conditions of this License
-Agreement.
-
-
-CNRI LICENSE AGREEMENT FOR PYTHON 1.6.1
----------------------------------------
-
-1. This LICENSE AGREEMENT is between the Corporation for National
-Research Initiatives, having an office at 1895 Preston White Drive,
-Reston, VA 20191 ("CNRI"), and the Individual or Organization
-("Licensee") accessing and otherwise using Python 1.6.1 software in
-source or binary form and its associated documentation.
-
-2. Subject to the terms and conditions of this License Agreement, CNRI
-hereby grants Licensee a nonexclusive, royalty-free, world-wide
-license to reproduce, analyze, test, perform and/or display publicly,
-prepare derivative works, distribute, and otherwise use Python 1.6.1
-alone or in any derivative version, provided, however, that CNRI's
-License Agreement and CNRI's notice of copyright, i.e., "Copyright (c)
-1995-2001 Corporation for National Research Initiatives; All Rights
-Reserved" are retained in Python 1.6.1 alone or in any derivative
-version prepared by Licensee. Alternately, in lieu of CNRI's License
-Agreement, Licensee may substitute the following text (omitting the
-quotes): "Python 1.6.1 is made available subject to the terms and
-conditions in CNRI's License Agreement. This Agreement together with
-Python 1.6.1 may be located on the Internet using the following
-unique, persistent identifier (known as a handle): 1895.22/1013. This
-Agreement may also be obtained from a proxy server on the Internet
-using the following URL: http://hdl.handle.net/1895.22/1013".
-
-3. In the event Licensee prepares a derivative work that is based on
-or incorporates Python 1.6.1 or any part thereof, and wants to make
-the derivative work available to others as provided herein, then
-Licensee hereby agrees to include in any such work a brief summary of
-the changes made to Python 1.6.1.
-
-4. CNRI is making Python 1.6.1 available to Licensee on an "AS IS"
-basis. CNRI MAKES NO REPRESENTATIONS OR WARRANTIES, EXPRESS OR
-IMPLIED. BY WAY OF EXAMPLE, BUT NOT LIMITATION, CNRI MAKES NO AND
-DISCLAIMS ANY REPRESENTATION OR WARRANTY OF MERCHANTABILITY OR FITNESS
-FOR ANY PARTICULAR PURPOSE OR THAT THE USE OF PYTHON 1.6.1 WILL NOT
-INFRINGE ANY THIRD PARTY RIGHTS.
-
-5. CNRI SHALL NOT BE LIABLE TO LICENSEE OR ANY OTHER USERS OF PYTHON
-1.6.1 FOR ANY INCIDENTAL, SPECIAL, OR CONSEQUENTIAL DAMAGES OR LOSS AS
-A RESULT OF MODIFYING, DISTRIBUTING, OR OTHERWISE USING PYTHON 1.6.1,
-OR ANY DERIVATIVE THEREOF, EVEN IF ADVISED OF THE POSSIBILITY THEREOF.
-
-6. This License Agreement will automatically terminate upon a material
-breach of its terms and conditions.
-
-7. This License Agreement shall be governed by the federal
-intellectual property law of the United States, including without
-limitation the federal copyright law, and, to the extent such
-U.S. federal law does not apply, by the law of the Commonwealth of
-Virginia, excluding Virginia's conflict of law provisions.
-Notwithstanding the foregoing, with regard to derivative works based
-on Python 1.6.1 that incorporate non-separable material that was
-previously distributed under the GNU General Public License (GPL), the
-law of the Commonwealth of Virginia shall govern this License
-Agreement only as to issues arising under or with respect to
-Paragraphs 4, 5, and 7 of this License Agreement. Nothing in this
-License Agreement shall be deemed to create any relationship of
-agency, partnership, or joint venture between CNRI and Licensee. This
-License Agreement does not grant permission to use CNRI trademarks or
-trade name in a trademark sense to endorse or promote products or
-services of Licensee, or any third party.
-
-8. By clicking on the "ACCEPT" button where indicated, or by copying,
-installing or otherwise using Python 1.6.1, Licensee agrees to be
-bound by the terms and conditions of this License Agreement.
-
- ACCEPT
-
-
-CWI LICENSE AGREEMENT FOR PYTHON 0.9.0 THROUGH 1.2
---------------------------------------------------
-
-Copyright (c) 1991 - 1995, Stichting Mathematisch Centrum Amsterdam,
-The Netherlands. All rights reserved.
-
-Permission to use, copy, modify, and distribute this software and its
-documentation for any purpose and without fee is hereby granted,
-provided that the above copyright notice appear in all copies and that
-both that copyright notice and this permission notice appear in
-supporting documentation, and that the name of Stichting Mathematisch
-Centrum or CWI not be used in advertising or publicity pertaining to
-distribution of the software without specific, written prior
-permission.
-
-STICHTING MATHEMATISCH CENTRUM DISCLAIMS ALL WARRANTIES WITH REGARD TO
-THIS SOFTWARE, INCLUDING ALL IMPLIED WARRANTIES OF MERCHANTABILITY AND
-FITNESS, IN NO EVENT SHALL STICHTING MATHEMATISCH CENTRUM BE LIABLE
-FOR ANY SPECIAL, INDIRECT OR CONSEQUENTIAL DAMAGES OR ANY DAMAGES
-WHATSOEVER RESULTING FROM LOSS OF USE, DATA OR PROFITS, WHETHER IN AN
-ACTION OF CONTRACT, NEGLIGENCE OR OTHER TORTIOUS ACTION, ARISING OUT
-OF OR IN CONNECTION WITH THE USE OR PERFORMANCE OF THIS SOFTWARE.
-
-This copy of Python includes a copy of bzip2, which is licensed under the following terms:
-
-
-This program, "bzip2", the associated library "libbzip2", and all
-documentation, are copyright (C) 1996-2005 Julian R Seward. All
-rights reserved.
-
-Redistribution and use in source and binary forms, with or without
-modification, are permitted provided that the following conditions
-are met:
-
-1. Redistributions of source code must retain the above copyright
- notice, this list of conditions and the following disclaimer.
-
-2. The origin of this software must not be misrepresented; you must
- not claim that you wrote the original software. If you use this
- software in a product, an acknowledgment in the product
- documentation would be appreciated but is not required.
-
-3. Altered source versions must be plainly marked as such, and must
- not be misrepresented as being the original software.
-
-4. The name of the author may not be used to endorse or promote
- products derived from this software without specific prior written
- permission.
-
-THIS SOFTWARE IS PROVIDED BY THE AUTHOR ``AS IS'' AND ANY EXPRESS
-OR IMPLIED WARRANTIES, INCLUDING, BUT NOT LIMITED TO, THE IMPLIED
-WARRANTIES OF MERCHANTABILITY AND FITNESS FOR A PARTICULAR PURPOSE
-ARE DISCLAIMED. IN NO EVENT SHALL THE AUTHOR BE LIABLE FOR ANY
-DIRECT, INDIRECT, INCIDENTAL, SPECIAL, EXEMPLARY, OR CONSEQUENTIAL
-DAMAGES (INCLUDING, BUT NOT LIMITED TO, PROCUREMENT OF SUBSTITUTE
-GOODS OR SERVICES; LOSS OF USE, DATA, OR PROFITS; OR BUSINESS
-INTERRUPTION) HOWEVER CAUSED AND ON ANY THEORY OF LIABILITY,
-WHETHER IN CONTRACT, STRICT LIABILITY, OR TORT (INCLUDING
-NEGLIGENCE OR OTHERWISE) ARISING IN ANY WAY OUT OF THE USE OF THIS
-SOFTWARE, EVEN IF ADVISED OF THE POSSIBILITY OF SUCH DAMAGE.
-
-Julian Seward, Cambridge, UK.
-jseward@acm.org
-bzip2/libbzip2 version 1.0.3 of 15 February 2005
-
-
-This copy of Python includes a copy of db, which is licensed under the following terms:
-
-/*-
- * $Id: LICENSE,v 12.1 2005/06/16 20:20:10 bostic Exp $
- */
-
-The following is the license that applies to this copy of the Berkeley DB
-software. For a license to use the Berkeley DB software under conditions
-other than those described here, or to purchase support for this software,
-please contact Sleepycat Software by email at info@sleepycat.com, or on
-the Web at http://www.sleepycat.com.
-
-=-=-=-=-=-=-=-=-=-=-=-=-=-=-=-=-=-=-=-=-=-=-=-=-=
-/*
- * Copyright (c) 1990-2005
- * Sleepycat Software. All rights reserved.
- *
- * Redistribution and use in source and binary forms, with or without
- * modification, are permitted provided that the following conditions
- * are met:
- * 1. Redistributions of source code must retain the above copyright
- * notice, this list of conditions and the following disclaimer.
- * 2. Redistributions in binary form must reproduce the above copyright
- * notice, this list of conditions and the following disclaimer in the
- * documentation and/or other materials provided with the distribution.
- * 3. Redistributions in any form must be accompanied by information on
- * how to obtain complete source code for the DB software and any
- * accompanying software that uses the DB software. The source code
- * must either be included in the distribution or be available for no
- * more than the cost of distribution plus a nominal fee, and must be
- * freely redistributable under reasonable conditions. For an
- * executable file, complete source code means the source code for all
- * modules it contains. It does not include source code for modules or
- * files that typically accompany the major components of the operating
- * system on which the executable file runs.
- *
- * THIS SOFTWARE IS PROVIDED BY SLEEPYCAT SOFTWARE ``AS IS'' AND ANY EXPRESS
- * OR IMPLIED WARRANTIES, INCLUDING, BUT NOT LIMITED TO, THE IMPLIED
- * WARRANTIES OF MERCHANTABILITY, FITNESS FOR A PARTICULAR PURPOSE, OR
- * NON-INFRINGEMENT, ARE DISCLAIMED. IN NO EVENT SHALL SLEEPYCAT SOFTWARE
- * BE LIABLE FOR ANY DIRECT, INDIRECT, INCIDENTAL, SPECIAL, EXEMPLARY, OR
- * CONSEQUENTIAL DAMAGES (INCLUDING, BUT NOT LIMITED TO, PROCUREMENT OF
- * SUBSTITUTE GOODS OR SERVICES; LOSS OF USE, DATA, OR PROFITS; OR BUSINESS
- * INTERRUPTION) HOWEVER CAUSED AND ON ANY THEORY OF LIABILITY, WHETHER IN
- * CONTRACT, STRICT LIABILITY, OR TORT (INCLUDING NEGLIGENCE OR OTHERWISE)
- * ARISING IN ANY WAY OUT OF THE USE OF THIS SOFTWARE, EVEN IF ADVISED OF
- * THE POSSIBILITY OF SUCH DAMAGE.
- */
-/*
- * Copyright (c) 1990, 1993, 1994, 1995
- * The Regents of the University of California. All rights reserved.
- *
- * Redistribution and use in source and binary forms, with or without
- * modification, are permitted provided that the following conditions
- * are met:
- * 1. Redistributions of source code must retain the above copyright
- * notice, this list of conditions and the following disclaimer.
- * 2. Redistributions in binary form must reproduce the above copyright
- * notice, this list of conditions and the following disclaimer in the
- * documentation and/or other materials provided with the distribution.
- * 3. Neither the name of the University nor the names of its contributors
- * may be used to endorse or promote products derived from this software
- * without specific prior written permission.
- *
- * THIS SOFTWARE IS PROVIDED BY THE REGENTS AND CONTRIBUTORS ``AS IS'' AND
- * ANY EXPRESS OR IMPLIED WARRANTIES, INCLUDING, BUT NOT LIMITED TO, THE
- * IMPLIED WARRANTIES OF MERCHANTABILITY AND FITNESS FOR A PARTICULAR PURPOSE
- * ARE DISCLAIMED. IN NO EVENT SHALL THE REGENTS OR CONTRIBUTORS BE LIABLE
- * FOR ANY DIRECT, INDIRECT, INCIDENTAL, SPECIAL, EXEMPLARY, OR CONSEQUENTIAL
- * DAMAGES (INCLUDING, BUT NOT LIMITED TO, PROCUREMENT OF SUBSTITUTE GOODS
- * OR SERVICES; LOSS OF USE, DATA, OR PROFITS; OR BUSINESS INTERRUPTION)
- * HOWEVER CAUSED AND ON ANY THEORY OF LIABILITY, WHETHER IN CONTRACT, STRICT
- * LIABILITY, OR TORT (INCLUDING NEGLIGENCE OR OTHERWISE) ARISING IN ANY WAY
- * OUT OF THE USE OF THIS SOFTWARE, EVEN IF ADVISED OF THE POSSIBILITY OF
- * SUCH DAMAGE.
- */
-/*
- * Copyright (c) 1995, 1996
- * The President and Fellows of Harvard University. All rights reserved.
- *
- * Redistribution and use in source and binary forms, with or without
- * modification, are permitted provided that the following conditions
- * are met:
- * 1. Redistributions of source code must retain the above copyright
- * notice, this list of conditions and the following disclaimer.
- * 2. Redistributions in binary form must reproduce the above copyright
- * notice, this list of conditions and the following disclaimer in the
- * documentation and/or other materials provided with the distribution.
- * 3. Neither the name of the University nor the names of its contributors
- * may be used to endorse or promote products derived from this software
- * without specific prior written permission.
- *
- * THIS SOFTWARE IS PROVIDED BY HARVARD AND ITS CONTRIBUTORS ``AS IS'' AND
- * ANY EXPRESS OR IMPLIED WARRANTIES, INCLUDING, BUT NOT LIMITED TO, THE
- * IMPLIED WARRANTIES OF MERCHANTABILITY AND FITNESS FOR A PARTICULAR PURPOSE
- * ARE DISCLAIMED. IN NO EVENT SHALL HARVARD OR ITS CONTRIBUTORS BE LIABLE
- * FOR ANY DIRECT, INDIRECT, INCIDENTAL, SPECIAL, EXEMPLARY, OR CONSEQUENTIAL
- * DAMAGES (INCLUDING, BUT NOT LIMITED TO, PROCUREMENT OF SUBSTITUTE GOODS
- * OR SERVICES; LOSS OF USE, DATA, OR PROFITS; OR BUSINESS INTERRUPTION)
- * HOWEVER CAUSED AND ON ANY THEORY OF LIABILITY, WHETHER IN CONTRACT, STRICT
- * LIABILITY, OR TORT (INCLUDING NEGLIGENCE OR OTHERWISE) ARISING IN ANY WAY
- * OUT OF THE USE OF THIS SOFTWARE, EVEN IF ADVISED OF THE POSSIBILITY OF
- * SUCH DAMAGE.
- */
-
-This copy of Python includes a copy of openssl, which is licensed under the following terms:
-
-
- LICENSE ISSUES
- ==============
-
- The OpenSSL toolkit stays under a dual license, i.e. both the conditions of
- the OpenSSL License and the original SSLeay license apply to the toolkit.
- See below for the actual license texts. Actually both licenses are BSD-style
- Open Source licenses. In case of any license issues related to OpenSSL
- please contact openssl-core@openssl.org.
-
- OpenSSL License
- ---------------
-
-/* ====================================================================
- * Copyright (c) 1998-2005 The OpenSSL Project. All rights reserved.
- *
- * Redistribution and use in source and binary forms, with or without
- * modification, are permitted provided that the following conditions
- * are met:
- *
- * 1. Redistributions of source code must retain the above copyright
- * notice, this list of conditions and the following disclaimer.
- *
- * 2. Redistributions in binary form must reproduce the above copyright
- * notice, this list of conditions and the following disclaimer in
- * the documentation and/or other materials provided with the
- * distribution.
- *
- * 3. All advertising materials mentioning features or use of this
- * software must display the following acknowledgment:
- * "This product includes software developed by the OpenSSL Project
- * for use in the OpenSSL Toolkit. (http://www.openssl.org/)"
- *
- * 4. The names "OpenSSL Toolkit" and "OpenSSL Project" must not be used to
- * endorse or promote products derived from this software without
- * prior written permission. For written permission, please contact
- * openssl-core@openssl.org.
- *
- * 5. Products derived from this software may not be called "OpenSSL"
- * nor may "OpenSSL" appear in their names without prior written
- * permission of the OpenSSL Project.
- *
- * 6. Redistributions of any form whatsoever must retain the following
- * acknowledgment:
- * "This product includes software developed by the OpenSSL Project
- * for use in the OpenSSL Toolkit (http://www.openssl.org/)"
- *
- * THIS SOFTWARE IS PROVIDED BY THE OpenSSL PROJECT ``AS IS'' AND ANY
- * EXPRESSED OR IMPLIED WARRANTIES, INCLUDING, BUT NOT LIMITED TO, THE
- * IMPLIED WARRANTIES OF MERCHANTABILITY AND FITNESS FOR A PARTICULAR
- * PURPOSE ARE DISCLAIMED. IN NO EVENT SHALL THE OpenSSL PROJECT OR
- * ITS CONTRIBUTORS BE LIABLE FOR ANY DIRECT, INDIRECT, INCIDENTAL,
- * SPECIAL, EXEMPLARY, OR CONSEQUENTIAL DAMAGES (INCLUDING, BUT
- * NOT LIMITED TO, PROCUREMENT OF SUBSTITUTE GOODS OR SERVICES;
- * LOSS OF USE, DATA, OR PROFITS; OR BUSINESS INTERRUPTION)
- * HOWEVER CAUSED AND ON ANY THEORY OF LIABILITY, WHETHER IN CONTRACT,
- * STRICT LIABILITY, OR TORT (INCLUDING NEGLIGENCE OR OTHERWISE)
- * ARISING IN ANY WAY OUT OF THE USE OF THIS SOFTWARE, EVEN IF ADVISED
- * OF THE POSSIBILITY OF SUCH DAMAGE.
- * ====================================================================
- *
- * This product includes cryptographic software written by Eric Young
- * (eay@cryptsoft.com). This product includes software written by Tim
- * Hudson (tjh@cryptsoft.com).
- *
- */
-
- Original SSLeay License
- -----------------------
-
-/* Copyright (C) 1995-1998 Eric Young (eay@cryptsoft.com)
- * All rights reserved.
- *
- * This package is an SSL implementation written
- * by Eric Young (eay@cryptsoft.com).
- * The implementation was written so as to conform with Netscapes SSL.
- *
- * This library is free for commercial and non-commercial use as long as
- * the following conditions are aheared to. The following conditions
- * apply to all code found in this distribution, be it the RC4, RSA,
- * lhash, DES, etc., code; not just the SSL code. The SSL documentation
- * included with this distribution is covered by the same copyright terms
- * except that the holder is Tim Hudson (tjh@cryptsoft.com).
- *
- * Copyright remains Eric Young's, and as such any Copyright notices in
- * the code are not to be removed.
- * If this package is used in a product, Eric Young should be given attribution
- * as the author of the parts of the library used.
- * This can be in the form of a textual message at program startup or
- * in documentation (online or textual) provided with the package.
- *
- * Redistribution and use in source and binary forms, with or without
- * modification, are permitted provided that the following conditions
- * are met:
- * 1. Redistributions of source code must retain the copyright
- * notice, this list of conditions and the following disclaimer.
- * 2. Redistributions in binary form must reproduce the above copyright
- * notice, this list of conditions and the following disclaimer in the
- * documentation and/or other materials provided with the distribution.
- * 3. All advertising materials mentioning features or use of this software
- * must display the following acknowledgement:
- * "This product includes cryptographic software written by
- * Eric Young (eay@cryptsoft.com)"
- * The word 'cryptographic' can be left out if the rouines from the library
- * being used are not cryptographic related :-).
- * 4. If you include any Windows specific code (or a derivative thereof) from
- * the apps directory (application code) you must include an acknowledgement:
- * "This product includes software written by Tim Hudson (tjh@cryptsoft.com)"
- *
- * THIS SOFTWARE IS PROVIDED BY ERIC YOUNG ``AS IS'' AND
- * ANY EXPRESS OR IMPLIED WARRANTIES, INCLUDING, BUT NOT LIMITED TO, THE
- * IMPLIED WARRANTIES OF MERCHANTABILITY AND FITNESS FOR A PARTICULAR PURPOSE
- * ARE DISCLAIMED. IN NO EVENT SHALL THE AUTHOR OR CONTRIBUTORS BE LIABLE
- * FOR ANY DIRECT, INDIRECT, INCIDENTAL, SPECIAL, EXEMPLARY, OR CONSEQUENTIAL
- * DAMAGES (INCLUDING, BUT NOT LIMITED TO, PROCUREMENT OF SUBSTITUTE GOODS
- * OR SERVICES; LOSS OF USE, DATA, OR PROFITS; OR BUSINESS INTERRUPTION)
- * HOWEVER CAUSED AND ON ANY THEORY OF LIABILITY, WHETHER IN CONTRACT, STRICT
- * LIABILITY, OR TORT (INCLUDING NEGLIGENCE OR OTHERWISE) ARISING IN ANY WAY
- * OUT OF THE USE OF THIS SOFTWARE, EVEN IF ADVISED OF THE POSSIBILITY OF
- * SUCH DAMAGE.
- *
- * The licence and distribution terms for any publically available version or
- * derivative of this code cannot be changed. i.e. this code cannot simply be
- * copied and put under another distribution licence
- * [including the GNU Public Licence.]
- */
-
-
-This copy of Python includes a copy of tcl, which is licensed under the following terms:
-
-This software is copyrighted by the Regents of the University of
-California, Sun Microsystems, Inc., Scriptics Corporation, ActiveState
-Corporation and other parties. The following terms apply to all files
-associated with the software unless explicitly disclaimed in
-individual files.
-
-The authors hereby grant permission to use, copy, modify, distribute,
-and license this software and its documentation for any purpose, provided
-that existing copyright notices are retained in all copies and that this
-notice is included verbatim in any distributions. No written agreement,
-license, or royalty fee is required for any of the authorized uses.
-Modifications to this software may be copyrighted by their authors
-and need not follow the licensing terms described here, provided that
-the new terms are clearly indicated on the first page of each file where
-they apply.
-
-IN NO EVENT SHALL THE AUTHORS OR DISTRIBUTORS BE LIABLE TO ANY PARTY
-FOR DIRECT, INDIRECT, SPECIAL, INCIDENTAL, OR CONSEQUENTIAL DAMAGES
-ARISING OUT OF THE USE OF THIS SOFTWARE, ITS DOCUMENTATION, OR ANY
-DERIVATIVES THEREOF, EVEN IF THE AUTHORS HAVE BEEN ADVISED OF THE
-POSSIBILITY OF SUCH DAMAGE.
-
-THE AUTHORS AND DISTRIBUTORS SPECIFICALLY DISCLAIM ANY WARRANTIES,
-INCLUDING, BUT NOT LIMITED TO, THE IMPLIED WARRANTIES OF MERCHANTABILITY,
-FITNESS FOR A PARTICULAR PURPOSE, AND NON-INFRINGEMENT. THIS SOFTWARE
-IS PROVIDED ON AN "AS IS" BASIS, AND THE AUTHORS AND DISTRIBUTORS HAVE
-NO OBLIGATION TO PROVIDE MAINTENANCE, SUPPORT, UPDATES, ENHANCEMENTS, OR
-MODIFICATIONS.
-
-GOVERNMENT USE: If you are acquiring this software on behalf of the
-U.S. government, the Government shall have only "Restricted Rights"
-in the software and related documentation as defined in the Federal
-Acquisition Regulations (FARs) in Clause 52.227.19 (c) (2). If you
-are acquiring the software on behalf of the Department of Defense, the
-software shall be classified as "Commercial Computer Software" and the
-Government shall have only "Restricted Rights" as defined in Clause
-252.227-7013 (c) (1) of DFARs. Notwithstanding the foregoing, the
-authors grant the U.S. Government and others acting in its behalf
-permission to use and distribute the software in accordance with the
-terms specified in this license.
-
-This copy of Python includes a copy of tk, which is licensed under the following terms:
-
-This software is copyrighted by the Regents of the University of
-California, Sun Microsystems, Inc., and other parties. The following
-terms apply to all files associated with the software unless explicitly
-disclaimed in individual files.
-
-The authors hereby grant permission to use, copy, modify, distribute,
-and license this software and its documentation for any purpose, provided
-that existing copyright notices are retained in all copies and that this
-notice is included verbatim in any distributions. No written agreement,
-license, or royalty fee is required for any of the authorized uses.
-Modifications to this software may be copyrighted by their authors
-and need not follow the licensing terms described here, provided that
-the new terms are clearly indicated on the first page of each file where
-they apply.
-
-IN NO EVENT SHALL THE AUTHORS OR DISTRIBUTORS BE LIABLE TO ANY PARTY
-FOR DIRECT, INDIRECT, SPECIAL, INCIDENTAL, OR CONSEQUENTIAL DAMAGES
-ARISING OUT OF THE USE OF THIS SOFTWARE, ITS DOCUMENTATION, OR ANY
-DERIVATIVES THEREOF, EVEN IF THE AUTHORS HAVE BEEN ADVISED OF THE
-POSSIBILITY OF SUCH DAMAGE.
-
-THE AUTHORS AND DISTRIBUTORS SPECIFICALLY DISCLAIM ANY WARRANTIES,
-INCLUDING, BUT NOT LIMITED TO, THE IMPLIED WARRANTIES OF MERCHANTABILITY,
-FITNESS FOR A PARTICULAR PURPOSE, AND NON-INFRINGEMENT. THIS SOFTWARE
-IS PROVIDED ON AN "AS IS" BASIS, AND THE AUTHORS AND DISTRIBUTORS HAVE
-NO OBLIGATION TO PROVIDE MAINTENANCE, SUPPORT, UPDATES, ENHANCEMENTS, OR
-MODIFICATIONS.
-
-GOVERNMENT USE: If you are acquiring this software on behalf of the
-U.S. government, the Government shall have only "Restricted Rights"
-in the software and related documentation as defined in the Federal
-Acquisition Regulations (FARs) in Clause 52.227.19 (c) (2). If you
-are acquiring the software on behalf of the Department of Defense, the
-software shall be classified as "Commercial Computer Software" and the
-Government shall have only "Restricted Rights" as defined in Clause
-252.227-7013 (c) (1) of DFARs. Notwithstanding the foregoing, the
-authors grant the U.S. Government and others acting in its behalf
-permission to use and distribute the software in accordance with the
-terms specified in this license.
diff --git a/src/extractcode/README.rst b/src/extractcode/README.rst
deleted file mode 100644
index bc5bfeb..0000000
--- a/src/extractcode/README.rst
+++ /dev/null
@@ -1,11 +0,0 @@
-extractcode is a universal archive extractor. It uses behind the scenes
-the Python standard library, a custom ctypes binding to libarchive and
-the 7zip command line to extract a large number of common and
-less common archives and compressed files. It tries to extract things
-in the same way on all OSes, including auto-renaming files that would
-not have valid names on certain filesystems or when there are multiple
-copies of the same path in a given archive.
-The extraction is driven from a "voting" system that considers the
-file extension(s) and name, the file type and mime type (using a ctypes
-binding to libmagic) to select the most appropriate extractor or
-uncompressor function. It can handle multi-level archives such as tar.gz.
diff --git a/src/extractcode/__init__.py b/src/extractcode/__init__.py
index b9900b4..b41efdc 100644
--- a/src/extractcode/__init__.py
+++ b/src/extractcode/__init__.py
@@ -1,30 +1,22 @@
#
-# Copyright (c) 2018 nexB Inc. and others. All rights reserved.
-# http://nexb.com and https://github.com/nexB/scancode-toolkit/
-# The ScanCode software is licensed under the Apache License version 2.0.
-# Data generated with ScanCode require an acknowledgment.
+# Copyright (c) nexB Inc. and others.
+# SPDX-License-Identifier: Apache-2.0
+#
+# Visit https://aboutcode.org and https://github.com/nexB/ for support and download.
# ScanCode is a trademark of nexB Inc.
#
-# You may not use this software except in compliance with the License.
-# You may obtain a copy of the License at: http://apache.org/licenses/LICENSE-2.0
-# Unless required by applicable law or agreed to in writing, software distributed
-# under the License is distributed on an "AS IS" BASIS, WITHOUT WARRANTIES OR
-# CONDITIONS OF ANY KIND, either express or implied. See the License for the
-# specific language governing permissions and limitations under the License.
+# Licensed under the Apache License, Version 2.0 (the "License");
+# you may not use this file except in compliance with the License.
+# You may obtain a copy of the License at
#
-# When you publish or redistribute any data created with ScanCode or any ScanCode
-# derivative work, you must accompany this data with the following acknowledgment:
+# http://www.apache.org/licenses/LICENSE-2.0
+#
+# Unless required by applicable law or agreed to in writing, software
+# distributed under the License is distributed on an "AS IS" BASIS,
+# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+# See the License for the specific language governing permissions and
+# limitations under the License.
#
-# Generated with ScanCode and provided on an "AS IS" BASIS, WITHOUT WARRANTIES
-# OR CONDITIONS OF ANY KIND, either express or implied. No content created from
-# ScanCode should be considered or used as legal advice. Consult an Attorney
-# for any legal advice.
-# ScanCode is a free software code scanning tool from nexB Inc. and others.
-# Visit https://github.com/nexB/scancode-toolkit/ for support and download.
-
-from __future__ import absolute_import
-from __future__ import print_function
-from __future__ import unicode_literals
import logging
import os
@@ -33,17 +25,16 @@
import shutil
import sys
+from os.path import dirname
+from os.path import join
+from os.path import exists
+
from commoncode.fileutils import as_posixpath
from commoncode.fileutils import create_dir
from commoncode.fileutils import file_name
-from commoncode.fileutils import fsencode
from commoncode.fileutils import parent_directory
from commoncode.text import toascii
from commoncode.system import on_linux
-from commoncode.system import py2
-from os.path import dirname
-from os.path import join
-from os.path import exists
logger = logging.getLogger(__name__)
DEBUG = False
@@ -53,16 +44,8 @@
root_dir = join(dirname(__file__), 'bin')
-POSIX_PATH_SEP = b'/' if (on_linux and py2) else '/'
-WIN_PATH_SEP = b'\\' if (on_linux and py2) else '\\'
-PATHS_SEPS = POSIX_PATH_SEP + WIN_PATH_SEP
-EMPTY_STRING = b'' if (on_linux and py2) else ''
-DOT = b'.' if (on_linux and py2) else '.'
-DOTDOT = DOT + DOT
-UNDERSCORE = b'_' if (on_linux and py2) else '_'
-
# Suffix added to extracted target_dir paths
-EXTRACT_SUFFIX = b'-extract' if (on_linux and py2) else r'-extract'
+EXTRACT_SUFFIX = '-extract'
# high level archive "kinds"
docs = 1
@@ -103,10 +86,7 @@ def is_extraction_path(path):
"""
Return True is the path points to an extraction path.
"""
- if on_linux and py2:
- path = fsencode(path)
-
- return path and path.rstrip(PATHS_SEPS).endswith(EXTRACT_SUFFIX)
+ return path and path.rstrip('\\/').endswith(EXTRACT_SUFFIX)
def is_extracted(location):
@@ -114,8 +94,6 @@ def is_extracted(location):
Return True is the location is already extracted to the corresponding
extraction location.
"""
- if on_linux and py2:
- location = fsencode(location)
return location and exists(get_extraction_path(location))
@@ -123,18 +101,14 @@ def get_extraction_path(path):
"""
Return a path where to extract.
"""
- if on_linux and py2:
- path = fsencode(path)
- return path.rstrip(PATHS_SEPS) + EXTRACT_SUFFIX
+ return path.rstrip('\\/') + EXTRACT_SUFFIX
def remove_archive_suffix(path):
"""
Remove all the extracted suffix from a path.
"""
- if on_linux and py2:
- path = fsencode(path)
- return re.sub(EXTRACT_SUFFIX, EMPTY_STRING, path)
+ return re.sub(EXTRACT_SUFFIX, '', path)
def remove_backslashes_and_dotdots(directory):
@@ -142,21 +116,16 @@ def remove_backslashes_and_dotdots(directory):
Walk a directory and rename the files if their names contain backslashes.
Return a list of errors if any.
"""
- if on_linux and py2:
- directory = fsencode(directory)
errors = []
for top, _, files in os.walk(directory):
for filename in files:
- if not (WIN_PATH_SEP in filename or DOTDOT in filename):
+ if not ('\\' in filename or '..' in filename):
continue
try:
- new_path = as_posixpath(filename)
- new_path = new_path.strip(POSIX_PATH_SEP)
- new_path = posixpath.normpath(new_path)
- new_path = new_path.replace(DOTDOT, POSIX_PATH_SEP)
- new_path = new_path.strip(POSIX_PATH_SEP)
+ new_path = as_posixpath(filename).strip('/')
+ new_path = posixpath.normpath(new_path).replace('..', '/').strip('/')
new_path = posixpath.normpath(new_path)
- segments = new_path.split(POSIX_PATH_SEP)
+ segments = new_path.split('/')
directory = join(top, *segments[:-1])
create_dir(directory)
shutil.move(join(top, filename), join(top, *segments))
@@ -180,9 +149,7 @@ def new_name(location, is_dir=False):
the extension unchanged.
"""
assert location
- if on_linux and py2:
- location = fsencode(location)
- location = location.rstrip(PATHS_SEPS)
+ location = location.rstrip('\\/')
assert location
parent = parent_directory(location)
@@ -193,8 +160,8 @@ def new_name(location, is_dir=False):
filename = file_name(location)
# corner case
- if filename in (DOT, DOT):
- filename = UNDERSCORE
+ if filename in ('.', '..'):
+ filename = '_'
# if unique, return this
if filename.lower() not in siblings_lower:
@@ -204,19 +171,19 @@ def new_name(location, is_dir=False):
if is_dir:
# directories do not have an "extension"
base_name = filename
- ext = EMPTY_STRING
+ ext = ''
else:
- base_name, dot, ext = filename.partition(DOT)
+ base_name, dot, ext = filename.partition('.')
if dot:
- ext = dot + ext
+ ext = f'.{ext}'
else:
base_name = filename
- ext = EMPTY_STRING
+ ext = ''
# find a unique filename, adding a counter int to the base_name
counter = 1
while 1:
- filename = base_name + UNDERSCORE + str(counter) + ext
+ filename = f'{base_name}_{counter}{ext}'
if filename.lower() not in siblings_lower:
break
counter += 1
diff --git a/src/extractcode/api.py b/src/extractcode/api.py
index 428b76c..a5bb86c 100644
--- a/src/extractcode/api.py
+++ b/src/extractcode/api.py
@@ -1,32 +1,22 @@
#
-# Copyright (c) nexB Inc. and others. All rights reserved.
-# http://nexb.com and https://github.com/nexB/scancode-toolkit/
-# The ScanCode software is licensed under the Apache License version 2.0.
-# Data generated with ScanCode require an acknowledgment.
+# Copyright (c) nexB Inc. and others.
+# SPDX-License-Identifier: Apache-2.0
+#
+# Visit https://aboutcode.org and https://github.com/nexB/ for support and download.
# ScanCode is a trademark of nexB Inc.
#
-# You may not use this software except in compliance with the License.
-# You may obtain a copy of the License at: http://apache.org/licenses/LICENSE-2.0
-# Unless required by applicable law or agreed to in writing, software distributed
-# under the License is distributed on an "AS IS" BASIS, WITHOUT WARRANTIES OR
-# CONDITIONS OF ANY KIND, either express or implied. See the License for the
-# specific language governing permissions and limitations under the License.
+# Licensed under the Apache License, Version 2.0 (the "License");
+# you may not use this file except in compliance with the License.
+# You may obtain a copy of the License at
#
-# When you publish or redistribute any data created with ScanCode or any ScanCode
-# derivative work, you must accompany this data with the following acknowledgment:
+# http://www.apache.org/licenses/LICENSE-2.0
+#
+# Unless required by applicable law or agreed to in writing, software
+# distributed under the License is distributed on an "AS IS" BASIS,
+# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+# See the License for the specific language governing permissions and
+# limitations under the License.
#
-# Generated with ScanCode and provided on an "AS IS" BASIS, WITHOUT WARRANTIES
-# OR CONDITIONS OF ANY KIND, either express or implied. No content created from
-# ScanCode should be considered or used as legal advice. Consult an Attorney
-# for any legal advice.
-# ScanCode is a free software code scanning tool from nexB Inc. and others.
-# Visit https://github.com/nexB/scancode-toolkit/ for support and download.
-
-from __future__ import absolute_import
-from __future__ import division
-from __future__ import print_function
-from __future__ import unicode_literals
-
"""
Note: this API is unstable and still evolving.
diff --git a/src/extractcode/archive.py b/src/extractcode/archive.py
index d3a8080..3946bb7 100644
--- a/src/extractcode/archive.py
+++ b/src/extractcode/archive.py
@@ -1,42 +1,32 @@
#
-# Copyright (c) 2018 nexB Inc. and others. All rights reserved.
-# http://nexb.com and https://github.com/nexB/scancode-toolkit/
-# The ScanCode software is licensed under the Apache License version 2.0.
-# Data generated with ScanCode require an acknowledgment.
+# Copyright (c) nexB Inc. and others.
+# SPDX-License-Identifier: Apache-2.0
+#
+# Visit https://aboutcode.org and https://github.com/nexB/ for support and download.
# ScanCode is a trademark of nexB Inc.
#
-# You may not use this software except in compliance with the License.
-# You may obtain a copy of the License at: http://apache.org/licenses/LICENSE-2.0
-# Unless required by applicable law or agreed to in writing, software distributed
-# under the License is distributed on an "AS IS" BASIS, WITHOUT WARRANTIES OR
-# CONDITIONS OF ANY KIND, either express or implied. See the License for the
-# specific language governing permissions and limitations under the License.
+# Licensed under the Apache License, Version 2.0 (the "License");
+# you may not use this file except in compliance with the License.
+# You may obtain a copy of the License at
#
-# When you publish or redistribute any data created with ScanCode or any ScanCode
-# derivative work, you must accompany this data with the following acknowledgment:
+# http://www.apache.org/licenses/LICENSE-2.0
+#
+# Unless required by applicable law or agreed to in writing, software
+# distributed under the License is distributed on an "AS IS" BASIS,
+# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+# See the License for the specific language governing permissions and
+# limitations under the License.
#
-# Generated with ScanCode and provided on an "AS IS" BASIS, WITHOUT WARRANTIES
-# OR CONDITIONS OF ANY KIND, either express or implied. No content created from
-# ScanCode should be considered or used as legal advice. Consult an Attorney
-# for any legal advice.
-# ScanCode is a free software code scanning tool from nexB Inc. and others.
-# Visit https://github.com/nexB/scancode-toolkit/ for support and download.
-
-from __future__ import absolute_import
-from __future__ import print_function
-from __future__ import unicode_literals
from collections import namedtuple
import logging
import os
-from commoncode import compat
from commoncode import fileutils
from commoncode import filetype
from commoncode import functional
from commoncode.ignore import is_ignored
from commoncode.system import on_linux
-from commoncode.system import py2
from typecode import contenttype
@@ -55,7 +45,6 @@
from extractcode.uncompress import uncompress_gzip
from extractcode.uncompress import uncompress_bzip2
-
logger = logging.getLogger(__name__)
TRACE = False
TRACE_DEEP = False
@@ -154,8 +143,6 @@ def get_best_handler(location, kinds=all_kinds):
"""
Return the best handler of None for the file at location.
"""
- if on_linux and py2:
- location = fileutils.fsencode(location)
location = os.path.abspath(os.path.expanduser(location))
if not filetype.is_file(location):
return
@@ -173,9 +160,6 @@ def get_handlers(location):
Return an iterable of (handler, type_matched, mime_matched,
extension_matched,) for this `location`.
"""
- if on_linux and py2:
- location = fileutils.fsencode(location)
-
if filetype.is_file(location):
T = contenttype.get_type(location)
@@ -197,8 +181,6 @@ def get_handlers(location):
mime_matched = handler.mimetypes and any(m in mtype for m in handler.mimetypes)
exts = handler.extensions
if exts:
- if on_linux and py2:
- exts = tuple(fileutils.fsencode(e) for e in exts)
extension_matched = exts and location.lower().endswith(exts)
if TRACE_DEEP:
@@ -326,13 +308,10 @@ def extract_twice(location, target_dir, extractor1, extractor2):
hard to trace and debug very quickly. A depth of two is simple and sane and
covers most common cases.
"""
- if on_linux and py2:
- location = fileutils.fsencode(location)
- target_dir = fileutils.fsencode(target_dir)
abs_location = os.path.abspath(os.path.expanduser(location))
- abs_target_dir = compat.unicode(os.path.abspath(os.path.expanduser(target_dir)))
+ abs_target_dir = str(os.path.abspath(os.path.expanduser(target_dir)))
# extract first the intermediate payload to a temp dir
- temp_target = compat.unicode(fileutils.get_temp_dir(prefix='extractcode-extract-'))
+ temp_target = str(fileutils.get_temp_dir(prefix='extractcode-extract-'))
warnings = extractor1(abs_location, temp_target)
if TRACE:
logger.debug('extract_twice: temp_target: %(temp_target)r' % locals())
@@ -364,9 +343,9 @@ def extract_with_fallback(location, target_dir, extractor1, extractor2):
and a fallback extractor will succeed.
"""
abs_location = os.path.abspath(os.path.expanduser(location))
- abs_target_dir = compat.unicode(os.path.abspath(os.path.expanduser(target_dir)))
+ abs_target_dir = str(os.path.abspath(os.path.expanduser(target_dir)))
# attempt extract first to a temp dir
- temp_target1 = compat.unicode(fileutils.get_temp_dir(prefix='extractcode-extract1-'))
+ temp_target1 = str(fileutils.get_temp_dir(prefix='extractcode-extract1-'))
try:
warnings = extractor1(abs_location, temp_target1)
if TRACE:
@@ -374,7 +353,7 @@ def extract_with_fallback(location, target_dir, extractor1, extractor2):
fileutils.copytree(temp_target1, abs_target_dir)
except:
try:
- temp_target2 = compat.unicode(fileutils.get_temp_dir(prefix='extractcode-extract2-'))
+ temp_target2 = str(fileutils.get_temp_dir(prefix='extractcode-extract2-'))
warnings = extractor2(abs_location, temp_target2)
if TRACE:
logger.debug('extract_with_fallback: temp_target2: %(temp_target2)r' % locals())
@@ -395,8 +374,8 @@ def try_to_extract(location, target_dir, extractor):
but do not care if this fails.
"""
abs_location = os.path.abspath(os.path.expanduser(location))
- abs_target_dir = compat.unicode(os.path.abspath(os.path.expanduser(target_dir)))
- temp_target = compat.unicode(fileutils.get_temp_dir(prefix='extractcode-extract1-'))
+ abs_target_dir = str(os.path.abspath(os.path.expanduser(target_dir)))
+ temp_target = str(fileutils.get_temp_dir(prefix='extractcode-extract1-'))
warnings = []
try:
warnings = extractor(abs_location, temp_target)
diff --git a/src/extractcode/cli.py b/src/extractcode/cli.py
index d59ddf7..5591e5d 100644
--- a/src/extractcode/cli.py
+++ b/src/extractcode/cli.py
@@ -1,50 +1,40 @@
+
+#
+# Copyright (c) nexB Inc. and others.
+# SPDX-License-Identifier: Apache-2.0
#
-# Copyright (c) 2018 nexB Inc. and others. All rights reserved.
-# http://nexb.com and https://github.com/nexB/scancode-toolkit/
-# The ScanCode software is licensed under the Apache License version 2.0.
-# Data generated with ScanCode require an acknowledgment.
+# Visit https://aboutcode.org and https://github.com/nexB/ for support and download.
# ScanCode is a trademark of nexB Inc.
#
-# You may not use this software except in compliance with the License.
-# You may obtain a copy of the License at: http://apache.org/licenses/LICENSE-2.0
-# Unless required by applicable law or agreed to in writing, software distributed
-# under the License is distributed on an "AS IS" BASIS, WITHOUT WARRANTIES OR
-# CONDITIONS OF ANY KIND, either express or implied. See the License for the
-# specific language governing permissions and limitations under the License.
+# Licensed under the Apache License, Version 2.0 (the "License");
+# you may not use this file except in compliance with the License.
+# You may obtain a copy of the License at
#
-# When you publish or redistribute any data created with ScanCode or any ScanCode
-# derivative work, you must accompany this data with the following acknowledgment:
+# http://www.apache.org/licenses/LICENSE-2.0
+#
+# Unless required by applicable law or agreed to in writing, software
+# distributed under the License is distributed on an "AS IS" BASIS,
+# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+# See the License for the specific language governing permissions and
+# limitations under the License.
#
-# Generated with ScanCode and provided on an "AS IS" BASIS, WITHOUT WARRANTIES
-# OR CONDITIONS OF ANY KIND, either express or implied. No content created from
-# ScanCode should be considered or used as legal advice. Consult an Attorney
-# for any legal advice.
-# ScanCode is a free software code scanning tool from nexB Inc. and others.
-# Visit https://github.com/nexB/scancode-toolkit/ for support and download.
-
-from __future__ import print_function
-from __future__ import absolute_import
-from __future__ import unicode_literals
-from functools import partial
-from os import path
+import os
+import functools
import click
click.disable_unicode_literals_warning = True
from commoncode import cliutils
-from commoncode import compat
from commoncode import fileutils
from commoncode import filetype
from commoncode.text import toascii
from extractcode.api import extract_archives
-
__version__ = '2020.09.21'
-
-echo_stderr = partial(click.secho, err=True)
+echo_stderr = functools.partial(click.secho, err=True)
def print_version(ctx, param, value):
@@ -53,26 +43,17 @@ def print_version(ctx, param, value):
echo_stderr('ExtractCode version ' + __version__)
ctx.exit()
+
info_text = '''
-ExtractCode is a mostly universal archive and compressed files extractor, with
+ExtractCode is a mostly universal archive and compressed files extractor, with
a particular focus on code archives.
-Visit https://github.com/nexB/scancode-toolkit/ for support and download.
+Visit https://aboutcode.org and https://github.com/nexB/extractcode/ for support and download.
'''
-notice_path = path.join(path.abspath(path.dirname(__file__)), 'NOTICE')
+notice_path = os.path.join(os.path.abspath(os.path.dirname(__file__)), 'NOTICE')
notice_text = open(notice_path).read()
-delimiter = '\n\n\n'
-[notice_text, extra_notice_text] = notice_text.split(delimiter, 1)
-extra_notice_text = delimiter + extra_notice_text
-
-delimiter = '\n\n '
-[notice_text, acknowledgment_text] = notice_text.split(delimiter, 1)
-acknowledgment_text = delimiter + acknowledgment_text
-
-notice = acknowledgment_text.strip().replace(' ', '')
-
def print_about(ctx, param, value):
"""
@@ -80,7 +61,7 @@ def print_about(ctx, param, value):
"""
if not value or ctx.resilient_parsing:
return
- click.echo(info_text + notice_text + acknowledgment_text + extra_notice_text)
+ click.echo(info_text + notice_text)
ctx.exit()
@@ -115,7 +96,7 @@ class ExtractCommand(cliutils.BaseCommand):
@click.command(name='extractcode', epilog=epilog_text, cls=ExtractCommand)
@click.pass_context
-@click.argument('input', metavar='', type=click.Path(exists=True, readable=True, path_type=fileutils.PATH_TYPE))
+@click.argument('input', metavar='', type=click.Path(exists=True, readable=True))
@click.option('--verbose', is_flag=True, default=False, help='Print verbose file-by-file progress messages.')
@click.option('--quiet', is_flag=True, default=False, help='Do not print any summary or progress message.')
@@ -130,11 +111,11 @@ def extractcode(ctx, input, verbose, quiet, shallow, replace_originals, ignore,
"""extract archives and compressed files found in the file or directory tree.
Archives found inside an extracted archive are extracted recursively.
- Extraction for each archive is done in-place in a new directory named
+ Extraction for each archive is done in-place in a new directory named
'-extract' created side-by-side with an archive.
"""
- abs_location = fileutils.as_posixpath(path.abspath(path.expanduser(input)))
+ abs_location = fileutils.as_posixpath(os.path.abspath(os.path.expanduser(input)))
def extract_event(item):
"""
@@ -145,7 +126,7 @@ def extract_event(item):
if not item:
return ''
source = item.source
- if not isinstance(source, compat.unicode):
+ if not isinstance(source, str):
source = toascii(source, translit=True).decode('utf-8', 'replace')
if verbose:
if item.done:
@@ -153,7 +134,7 @@ def extract_event(item):
line = source and get_relative_path(path=source, len_base_path=len_base_path, base_is_dir=base_is_dir) or ''
else:
line = source and fileutils.file_name(source) or ''
- if not isinstance(line, compat.unicode):
+ if not isinstance(line, str):
line = toascii(line, translit=True).decode('utf-8', 'replace')
return 'Extracting: %(line)s' % locals()
@@ -168,7 +149,7 @@ def display_extract_summary():
has_errors = has_errors or bool(xev.errors)
has_warnings = has_warnings or bool(xev.warnings)
source = fileutils.as_posixpath(xev.source)
- if not isinstance(source, compat.unicode):
+ if not isinstance(source, str):
source = toascii(source, translit=True).decode('utf-8', 'replace')
source = get_relative_path(path=source, len_base_path=len_base_path, base_is_dir=base_is_dir)
for e in xev.errors:
@@ -191,7 +172,7 @@ def display_extract_summary():
extract_result_with_errors = []
unique_extract_events_with_errors = set()
has_extract_errors = False
-
+
extractibles = extract_archives(
abs_location, recurse=not shallow, replace_originals=replace_originals, ignore_pattern=ignore)
@@ -222,7 +203,7 @@ def get_relative_path(path, len_base_path, base_is_dir):
base path of `len_base_path` length where the base is a directory if
`base_is_dir` True or a file otherwise.
"""
- path = fileutils.fsdecode(path)
+ path = os.fsdecode(path)
if base_is_dir:
rel_path = path[len_base_path:]
else:
diff --git a/src/extractcode/extract.py b/src/extractcode/extract.py
index b08f90e..21a6e33 100644
--- a/src/extractcode/extract.py
+++ b/src/extractcode/extract.py
@@ -1,43 +1,35 @@
#
-# Copyright (c) 2018 nexB Inc. and others. All rights reserved.
-# http://nexb.com and https://github.com/nexB/scancode-toolkit/
-# The ScanCode software is licensed under the Apache License version 2.0.
-# Data generated with ScanCode require an acknowledgment.
+# Copyright (c) nexB Inc. and others.
+# SPDX-License-Identifier: Apache-2.0
+#
+# Visit https://aboutcode.org and https://github.com/nexB/ for support and download.
# ScanCode is a trademark of nexB Inc.
#
-# You may not use this software except in compliance with the License.
-# You may obtain a copy of the License at: http://apache.org/licenses/LICENSE-2.0
-# Unless required by applicable law or agreed to in writing, software distributed
-# under the License is distributed on an "AS IS" BASIS, WITHOUT WARRANTIES OR
-# CONDITIONS OF ANY KIND, either express or implied. See the License for the
-# specific language governing permissions and limitations under the License.
+# Licensed under the Apache License, Version 2.0 (the "License");
+# you may not use this file except in compliance with the License.
+# You may obtain a copy of the License at
+#
+# http://www.apache.org/licenses/LICENSE-2.0
#
-# When you publish or redistribute any data created with ScanCode or any ScanCode
-# derivative work, you must accompany this data with the following acknowledgment:
+# Unless required by applicable law or agreed to in writing, software
+# distributed under the License is distributed on an "AS IS" BASIS,
+# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+# See the License for the specific language governing permissions and
+# limitations under the License.
#
-# Generated with ScanCode and provided on an "AS IS" BASIS, WITHOUT WARRANTIES
-# OR CONDITIONS OF ANY KIND, either express or implied. No content created from
-# ScanCode should be considered or used as legal advice. Consult an Attorney
-# for any legal advice.
-# ScanCode is a free software code scanning tool from nexB Inc. and others.
-# Visit https://github.com/nexB/scancode-toolkit/ for support and download.
-from __future__ import absolute_import
-from __future__ import print_function
-from __future__ import unicode_literals
+import logging
+import traceback
from collections import namedtuple
from functools import partial
-import logging
from os.path import abspath
from os.path import expanduser
from os.path import join
-import traceback
from commoncode import fileutils
from commoncode import ignore
import extractcode
-from extractcode import archive
logger = logging.getLogger(__name__)
TRACE = False
@@ -189,7 +181,7 @@ def extract_files(
logger.debug('extract:walk not recurse: skipped file: %(loc)r' % locals())
continue
- if not archive.should_extract(loc, kinds, ignore_pattern):
+ if not extractcode.archive.should_extract(loc, kinds, ignore_pattern):
if TRACE:
logger.debug('extract:walk: skipped file: not should_extract: %(loc)r' % locals())
continue
@@ -219,9 +211,9 @@ def extract_files(
def extract_file(
- location,
- target,
- kinds=extractcode.default_kinds,
+ location,
+ target,
+ kinds=extractcode.default_kinds,
verbose=False,
):
"""
@@ -230,13 +222,11 @@ def extract_file(
"""
warnings = []
errors = []
- extractor = archive.get_extractor(location, kinds)
+ extractor = extractcode.archive.get_extractor(location, kinds)
if TRACE:
emodule = getattr(extractor, '__module__', '')
ename = getattr(extractor, '__name__', '')
- logger.debug(
- 'extract_file: extractor: for: {location} with kinds: {kinds}: {emodule}.{ename}'
- .format(**locals()))
+ logger.debug(f'extract_file: extractor: for: {location} with kinds: {kinds}: {emodule}.{ename}')
if extractor:
yield ExtractEvent(location, target, done=False, warnings=[], errors=[])
diff --git a/src/extractcode/libarchive2.py b/src/extractcode/libarchive2.py
index 70eeeb5..6b72456 100644
--- a/src/extractcode/libarchive2.py
+++ b/src/extractcode/libarchive2.py
@@ -1,31 +1,22 @@
#
-# Copyright (c) 2018 nexB Inc. and others. All rights reserved.
-# http://nexb.com and https://github.com/nexB/scancode-toolkit/
-# The ScanCode software is licensed under the Apache License version 2.0.
-# Data generated with ScanCode require an acknowledgment.
+# Copyright (c) nexB Inc. and others.
+# SPDX-License-Identifier: Apache-2.0
+#
+# Visit https://aboutcode.org and https://github.com/nexB/ for support and download.
# ScanCode is a trademark of nexB Inc.
#
-# You may not use this software except in compliance with the License.
-# You may obtain a copy of the License at: http://apache.org/licenses/LICENSE-2.0
-# Unless required by applicable law or agreed to in writing, software distributed
-# under the License is distributed on an "AS IS" BASIS, WITHOUT WARRANTIES OR
-# CONDITIONS OF ANY KIND, either express or implied. See the License for the
-# specific language governing permissions and limitations under the License.
+# Licensed under the Apache License, Version 2.0 (the "License");
+# you may not use this file except in compliance with the License.
+# You may obtain a copy of the License at
+#
+# http://www.apache.org/licenses/LICENSE-2.0
#
-# When you publish or redistribute any data created with ScanCode or any ScanCode
-# derivative work, you must accompany this data with the following acknowledgment:
+# Unless required by applicable law or agreed to in writing, software
+# distributed under the License is distributed on an "AS IS" BASIS,
+# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+# See the License for the specific language governing permissions and
+# limitations under the License.
#
-# Generated with ScanCode and provided on an "AS IS" BASIS, WITHOUT WARRANTIES
-# OR CONDITIONS OF ANY KIND, either express or implied. No content created from
-# ScanCode should be considered or used as legal advice. Consult an Attorney
-# for any legal advice.
-# ScanCode is a free software code scanning tool from nexB Inc. and others.
-# Visit https://github.com/nexB/scancode-toolkit/ for support and download.
-
-from __future__ import absolute_import
-from __future__ import division
-from __future__ import print_function
-from __future__ import unicode_literals
from functools import partial
import locale
@@ -43,11 +34,8 @@
import attr
from commoncode import command
-from commoncode import compat
from commoncode import fileutils
from commoncode import paths
-from commoncode.system import py2
-from commoncode.system import py3
from commoncode import text
import extractcode
@@ -262,7 +250,7 @@ def close(self):
free_archive(self.archive_struct)
self.archive_struct = None
- def iter(self, verbose=False):
+ def iter(self):
"""
Yield Entry for this archive.
"""
@@ -387,13 +375,7 @@ def get_path(self, func, func_w):
path = func(self.entry_struct)
if not path:
path = func_w(self.entry_struct)
-
- if py2 and isinstance(path, compat.unicode):
- # FIXME: encoding MAY fail if the encoding is NOT UTF-8!
- # .... should we transliterate there?
- path = path.encode('utf-8')
-
- if py3 and not isinstance(path, compat.unicode):
+ if not isinstance(path, str):
path = text.as_unicode(path)
return path
diff --git a/src/extractcode/patch.py b/src/extractcode/patch.py
index da8ab59..8a197ca 100644
--- a/src/extractcode/patch.py
+++ b/src/extractcode/patch.py
@@ -1,30 +1,22 @@
#
-# Copyright (c) 2015 nexB Inc. and others. All rights reserved.
-# http://nexb.com and https://github.com/nexB/scancode-toolkit/
-# The ScanCode software is licensed under the Apache License version 2.0.
-# Data generated with ScanCode require an acknowledgment.
+# Copyright (c) nexB Inc. and others.
+# SPDX-License-Identifier: Apache-2.0
+#
+# Visit https://aboutcode.org and https://github.com/nexB/ for support and download.
# ScanCode is a trademark of nexB Inc.
#
-# You may not use this software except in compliance with the License.
-# You may obtain a copy of the License at: http://apache.org/licenses/LICENSE-2.0
-# Unless required by applicable law or agreed to in writing, software distributed
-# under the License is distributed on an "AS IS" BASIS, WITHOUT WARRANTIES OR
-# CONDITIONS OF ANY KIND, either express or implied. See the License for the
-# specific language governing permissions and limitations under the License.
+# Licensed under the Apache License, Version 2.0 (the "License");
+# you may not use this file except in compliance with the License.
+# You may obtain a copy of the License at
#
-# When you publish or redistribute any data created with ScanCode or any ScanCode
-# derivative work, you must accompany this data with the following acknowledgment:
+# http://www.apache.org/licenses/LICENSE-2.0
+#
+# Unless required by applicable law or agreed to in writing, software
+# distributed under the License is distributed on an "AS IS" BASIS,
+# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+# See the License for the specific language governing permissions and
+# limitations under the License.
#
-# Generated with ScanCode and provided on an "AS IS" BASIS, WITHOUT WARRANTIES
-# OR CONDITIONS OF ANY KIND, either express or implied. No content created from
-# ScanCode should be considered or used as legal advice. Consult an Attorney
-# for any legal advice.
-# ScanCode is a free software code scanning tool from nexB Inc. and others.
-# Visit https://github.com/nexB/scancode-toolkit/ for support and download.
-
-from __future__ import absolute_import
-from __future__ import print_function
-from __future__ import unicode_literals
import posixpath
import logging
@@ -34,13 +26,11 @@
from commoncode import paths
from commoncode import fileutils
-from commoncode.system import py2
-from commoncode.system import py3
from commoncode import text
-import extractcode
-from extractcode import ExtractErrorFailedToExtract
import typecode.contenttype
+import extractcode
+from extractcode import ExtractErrorFailedToExtract
"""
Low level utilities to parse patch files and treat them as if they were
@@ -93,15 +83,9 @@ def extract(location, target_dir):
# write the location proper, with a suffix extension to avoid
# recursive extraction
- if py2:
- mode = 'wb'
- eol = b'\n'
- if py3:
- mode = 'w'
- eol = u'\n'
subfile_path = base_subfile_path + extractcode.EXTRACT_SUFFIX
- with open(subfile_path, mode) as subfile:
- subfile.write(eol.join(text))
+ with open(subfile_path, 'w') as subfile:
+ subfile.write('\n'.join(text))
return []
@@ -113,9 +97,11 @@ def is_patch(location, include_extracted=False):
"""
T = typecode.contenttype.get_type(location)
file_name = fileutils.file_name(location)
- patch_like = ('diff ' in T.filetype_file.lower()
- or '.diff' in file_name
- or '.patch' in file_name)
+ patch_like = (
+ 'diff ' in T.filetype_file.lower()
+ or '.diff' in file_name
+ or '.patch' in file_name
+ )
if not patch_like:
return False
diff --git a/src/extractcode/sevenzip.py b/src/extractcode/sevenzip.py
index ebc8498..b34bfbd 100644
--- a/src/extractcode/sevenzip.py
+++ b/src/extractcode/sevenzip.py
@@ -1,30 +1,23 @@
+
+#
+# Copyright (c) nexB Inc. and others.
+# SPDX-License-Identifier: Apache-2.0
#
-# Copyright (c) nexB Inc. and others. All rights reserved.
-# http://nexb.com and https://github.com/nexB/scancode-toolkit/
-# The ScanCode software is licensed under the Apache License version 2.0.
-# Data generated with ScanCode require an acknowledgment.
+# Visit https://aboutcode.org and https://github.com/nexB/ for support and download.
# ScanCode is a trademark of nexB Inc.
#
-# You may not use this software except in compliance with the License.
-# You may obtain a copy of the License at: http://apache.org/licenses/LICENSE-2.0
-# Unless required by applicable law or agreed to in writing, software distributed
-# under the License is distributed on an "AS IS" BASIS, WITHOUT WARRANTIES OR
-# CONDITIONS OF ANY KIND, either express or implied. See the License for the
-# specific language governing permissions and limitations under the License.
+# Licensed under the Apache License, Version 2.0 (the "License");
+# you may not use this file except in compliance with the License.
+# You may obtain a copy of the License at
#
-# When you publish or redistribute any data created with ScanCode or any ScanCode
-# derivative work, you must accompany this data with the following acknowledgment:
+# http://www.apache.org/licenses/LICENSE-2.0
+#
+# Unless required by applicable law or agreed to in writing, software
+# distributed under the License is distributed on an "AS IS" BASIS,
+# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+# See the License for the specific language governing permissions and
+# limitations under the License.
#
-# Generated with ScanCode and provided on an "AS IS" BASIS, WITHOUT WARRANTIES
-# OR CONDITIONS OF ANY KIND, either express or implied. No content created from
-# ScanCode should be considered or used as legal advice. Consult an Attorney
-# for any legal advice.
-# ScanCode is a free software code scanning tool from nexB Inc. and others.
-# Visit https://github.com/nexB/scancode-toolkit/ for support and download.
-
-from __future__ import absolute_import
-from __future__ import print_function
-from __future__ import unicode_literals
from collections import defaultdict
import io
@@ -42,18 +35,13 @@
from commoncode.system import on_mac
from commoncode.system import on_macos_14_or_higher
from commoncode.system import on_windows
-from commoncode.system import py3
from commoncode import text
import extractcode
from extractcode import ExtractErrorFailedToExtract
from extractcode import ExtractWarningIncorrectEntry
-if py3:
- from shlex import quote as shlex_quote # NOQA
-else:
- from pipes import quote as shlex_quote # NOQA
-
+from shlex import quote as shlex_quote
"""
Low level support for p/7zip-based archive extraction.
@@ -95,8 +83,12 @@ def get_7z_errors(stdout, stderr):
if not stdout or not stdout.strip():
return
- # ERROR: Can not create symbolic link : A required privilege is not held by the client. : .\2-SYMTYPE
- find_7z_errors = re.compile('^Error:(.*)$', re.MULTILINE | re.DOTALL | re.IGNORECASE).findall # NOQA
+ # ERROR: Can not create symbolic link : A required privilege is not held by
+ # the client. : .\2-SYMTYPE
+ find_7z_errors = re.compile(
+ '^Error:(.*)$',
+ re.MULTILINE | re.DOTALL | re.IGNORECASE
+ ).findall
stdlow = stderr.lower()
for err, msg in sevenzip_errors:
@@ -558,29 +550,11 @@ def parse_7z_listing(location, utf=False):
We ignore the header and footer in a listing.
"""
- if utf or py3:
- # read to unicode
- with io.open(location, 'r', encoding='utf-8') as listing:
- text = listing.read()
- text = text.replace(u'\r\n', u'\n')
-
- end_of_header = u'----------\n'
- path_key = u'Path'
- kv_sep = u'='
- path_blocks_sep = u'\n\n'
- line_sep = u'\n'
-
- else:
- # read to bytes
- with io.open(location, 'rb') as listing:
- text = listing.read()
- text = text.replace(b'\r\n', b'\n')
-
- end_of_header = b'----------\n'
- path_key = b'Path'
- kv_sep = b'='
- path_blocks_sep = b'\n\n'
- line_sep = b'\n'
+ # read to unicode
+ with io.open(location, 'r', encoding='utf-8') as listing:
+ text = listing.read()
+ # normalize line endings to POSIX
+ text = text.replace('\r\n', '\n')
if TRACE:
logger.debug('parse_7z_listing: initial text: type: ' + repr(type(text)))
@@ -588,7 +562,8 @@ def parse_7z_listing(location, utf=False):
print(text)
print('--------------------------------------')
- # for now we ignore the header
+ # for now we ignore the header, and only start dealing with text after that
+ end_of_header = '----------\n'
_header, _, paths = text.rpartition(end_of_header)
if not paths:
@@ -601,8 +576,11 @@ def parse_7z_listing(location, utf=False):
# (unless there is a \n in file name which is an error condition)
# - ends with an empty line
# then we have a global footer
+ two_empty_lines = '\n\n'
+ path_key = 'Path'
+ path_blocks = [pb for pb in paths.split(two_empty_lines) if pb and path_key in pb]
- path_blocks = [pb for pb in paths.split(path_blocks_sep) if pb and path_key in pb]
+ key_value_sep = '='
entries = []
@@ -613,22 +591,22 @@ def parse_7z_listing(location, utf=False):
continue
# we have a weird case of path with line returns in the file name
# we concatenate these in the first Path line
- while len(lines) > 1 and lines[0].startswith(path_key) and kv_sep not in lines[1]:
+ while len(lines) > 1 and lines[0].startswith(path_key) and key_value_sep not in lines[1]:
first_line = lines[0]
second_line = lines.pop(1)
- first_line = line_sep.join([first_line, second_line])
+ first_line = '\n'.join([first_line, second_line])
lines[0] = first_line
- dangling_lines = [line for line in lines if kv_sep not in line]
+ dangling_lines = [line for line in lines if key_value_sep not in line]
entry_errors = []
if dangling_lines:
emsg = 'Invalid 7z listing path block missing "=" as key/value separator: {}'.format(repr(path_block))
entry_errors.append(emsg)
entry_attributes = {}
- key_lines = [line for line in lines if kv_sep in line]
+ key_lines = [line for line in lines if key_value_sep in line]
for line in key_lines:
- k, _, v = line.partition(kv_sep)
+ k, _, v = line.partition(key_value_sep)
k = k.strip()
v = v.strip()
entry_attributes[k] = v
@@ -643,15 +621,6 @@ def parse_7z_listing(location, utf=False):
return entries
-def filter_entries(entries):
- """
- Given an iterable of entries, return two list of entries:
- a list of valid entries that can be extracted and a list entries that cannot
- be extracted.
- """
- # extractible
-
-
@attr.s(slots=True)
class Entry(object):
"""
diff --git a/src/extractcode/uncompress.py b/src/extractcode/uncompress.py
index 4dcc2eb..f584e84 100644
--- a/src/extractcode/uncompress.py
+++ b/src/extractcode/uncompress.py
@@ -1,50 +1,34 @@
#
-# Copyright (c) 2018 nexB Inc. and others. All rights reserved.
-# http://nexb.com and https://github.com/nexB/scancode-toolkit/
-# The ScanCode software is licensed under the Apache License version 2.0.
-# Data generated with ScanCode require an acknowledgment.
+# Copyright (c) nexB Inc. and others.
+# SPDX-License-Identifier: Apache-2.0
+#
+# Visit https://aboutcode.org and https://github.com/nexB/ for support and download.
# ScanCode is a trademark of nexB Inc.
#
-# You may not use this software except in compliance with the License.
-# You may obtain a copy of the License at: http://apache.org/licenses/LICENSE-2.0
-# Unless required by applicable law or agreed to in writing, software distributed
-# under the License is distributed on an "AS IS" BASIS, WITHOUT WARRANTIES OR
-# CONDITIONS OF ANY KIND, either express or implied. See the License for the
-# specific language governing permissions and limitations under the License.
+# Licensed under the Apache License, Version 2.0 (the "License");
+# you may not use this file except in compliance with the License.
+# You may obtain a copy of the License at
#
-# When you publish or redistribute any data created with ScanCode or any ScanCode
-# derivative work, you must accompany this data with the following acknowledgment:
+# http://www.apache.org/licenses/LICENSE-2.0
+#
+# Unless required by applicable law or agreed to in writing, software
+# distributed under the License is distributed on an "AS IS" BASIS,
+# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+# See the License for the specific language governing permissions and
+# limitations under the License.
#
-# Generated with ScanCode and provided on an "AS IS" BASIS, WITHOUT WARRANTIES
-# OR CONDITIONS OF ANY KIND, either express or implied. No content created from
-# ScanCode should be considered or used as legal advice. Consult an Attorney
-# for any legal advice.
-# ScanCode is a free software code scanning tool from nexB Inc. and others.
-# Visit https://github.com/nexB/scancode-toolkit/ for support and download.
-
-from __future__ import absolute_import
-from __future__ import print_function
-from __future__ import unicode_literals
-from functools import partial
+import bz2
import gzip
import logging
import os
import shutil
-try:
- # These imports add support for multistream BZ2 files
- # This is a Python2 backport for bz2file from Python3
- # Because of http://bugs.python.org/issue20781
- from bz2file import BZ2File
-except ImportError:
- from bz2 import BZ2File
-
+from functools import partial
from commoncode import fileutils
-from commoncode.system import py2
-from extractcode import EXTRACT_SUFFIX
+from extractcode import EXTRACT_SUFFIX
DEBUG = False
logger = logging.getLogger(__name__)
@@ -65,7 +49,9 @@ def uncompress(location, target_dir, decompressor, suffix=EXTRACT_SUFFIX):
# name when present.
if DEBUG:
logger.debug('uncompress: ' + location)
+
tmp_loc, warnings = uncompress_file(location, decompressor)
+
target_location = os.path.join(target_dir, os.path.basename(location) + suffix)
if os.path.exists(target_location):
fileutils.delete(target_location)
@@ -87,7 +73,9 @@ def uncompress_file(location, decompressor):
warnings = []
base_name = fileutils.file_base_name(location)
- target_location = os.path.join(fileutils.get_temp_dir(prefix='extractcode-extract-'), base_name)
+ target_location = os.path.join(fileutils.get_temp_dir(
+ prefix='extractcode-extract-'), base_name)
+
with decompressor(location, 'rb') as compressed:
with open(target_location, 'wb') as uncompressed:
buffer_size = 32 * 1024 * 1024
@@ -96,8 +84,10 @@ def uncompress_file(location, decompressor):
if not chunk:
break
uncompressed.write(chunk)
+
if getattr(decompressor, 'has_trailing_garbage', False):
warnings.append(location + ': Trailing garbage found and ignored.')
+
return target_location, warnings
@@ -106,7 +96,7 @@ def uncompress_bzip2(location, target_dir):
Uncompress a bzip2 compressed file at location in the target_dir.
Return a list warnings messages.
"""
- return uncompress(location, target_dir, BZ2File)
+ return uncompress(location, target_dir, decompressor=bz2.BZ2File)
def uncompress_gzip(location, target_dir):
@@ -115,38 +105,7 @@ def uncompress_gzip(location, target_dir):
Return a list warnings messages.
"""
- return uncompress(location, target_dir, GzipFileWithTrailing)
-
-
-class _GzipFileWithTrailing(gzip.GzipFile):
- """
- A subclass of gzip.GzipFile supporting files with trailing garbage. Ignore
- the garbage.
- """
- # TODO: what is first_file??
- first_file = True
- gzip_magic = b'\037\213'
- has_trailing_garbage = False
-
- def _read_gzip_header(self):
- # read the first two bytes
- magic = self.fileobj.read(2)
- # rewind two bytes back
- self.fileobj.seek(-2, os.SEEK_CUR)
- is_gzip = magic != self.gzip_magic
- if is_gzip and not self.first_file:
- self.first_file = False
- self.has_trailing_garbage = True
- raise EOFError('Trailing garbage found')
-
- self.first_file = False
- gzip.GzipFile._read_gzip_header(self)
-
-if py2:
- GzipFileWithTrailing = _GzipFileWithTrailing
-else:
- # FIXME: there is no easy way to monkey patch the gzip.py code in Python 3
- GzipFileWithTrailing = gzip.GzipFile
+ return uncompress(location, target_dir, decompressor=gzip.GzipFile)
def get_compressed_file_content(location, decompressor):
@@ -163,5 +122,5 @@ def get_compressed_file_content(location, decompressor):
return content, warnings
-get_gz_compressed_file_content = partial(get_compressed_file_content, decompressor=GzipFileWithTrailing)
-get_bz2_compressed_file_content = partial(get_compressed_file_content, decompressor=BZ2File)
+get_gz_compressed_file_content = partial(get_compressed_file_content, decompressor=gzip.GzipFile)
+get_bz2_compressed_file_content = partial(get_compressed_file_content, decompressor=bz2.BZ2File)
diff --git a/tests/README.rst b/tests/README.rst
new file mode 100644
index 0000000..d94783e
--- /dev/null
+++ b/tests/README.rst
@@ -0,0 +1,2 @@
+Put your Python test modules in this directory.
+
diff --git a/tests/extractcode/extractcode_assert_utils.py b/tests/extractcode/extractcode_assert_utils.py
index 786c90c..537cf1d 100644
--- a/tests/extractcode/extractcode_assert_utils.py
+++ b/tests/extractcode/extractcode_assert_utils.py
@@ -1,43 +1,33 @@
+
+#
+# Copyright (c) nexB Inc. and others.
+# SPDX-License-Identifier: Apache-2.0
#
-# Copyright (c) 2015 nexB Inc. and others. All rights reserved.
-# http://nexb.com and https://github.com/nexB/scancode-toolkit/
-# The ScanCode software is licensed under the Apache License version 2.0.
-# Data generated with ScanCode require an acknowledgment.
+# Visit https://aboutcode.org and https://github.com/nexB/ for support and download.
# ScanCode is a trademark of nexB Inc.
#
-# You may not use this software except in compliance with the License.
-# You may obtain a copy of the License at: http://apache.org/licenses/LICENSE-2.0
-# Unless required by applicable law or agreed to in writing, software distributed
-# under the License is distributed on an "AS IS" BASIS, WITHOUT WARRANTIES OR
-# CONDITIONS OF ANY KIND, either express or implied. See the License for the
-# specific language governing permissions and limitations under the License.
+# Licensed under the Apache License, Version 2.0 (the "License");
+# you may not use this file except in compliance with the License.
+# You may obtain a copy of the License at
#
-# When you publish or redistribute any data created with ScanCode or any ScanCode
-# derivative work, you must accompany this data with the following acknowledgment:
+# http://www.apache.org/licenses/LICENSE-2.0
+#
+# Unless required by applicable law or agreed to in writing, software
+# distributed under the License is distributed on an "AS IS" BASIS,
+# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+# See the License for the specific language governing permissions and
+# limitations under the License.
#
-# Generated with ScanCode and provided on an "AS IS" BASIS, WITHOUT WARRANTIES
-# OR CONDITIONS OF ANY KIND, either express or implied. No content created from
-# ScanCode should be considered or used as legal advice. Consult an Attorney
-# for any legal advice.
-# ScanCode is a free software code scanning tool from nexB Inc. and others.
-# Visit https://github.com/nexB/scancode-toolkit/ for support and download.
-
-from __future__ import absolute_import
-from __future__ import print_function
-from collections import OrderedDict
import json
import os
import ntpath
import posixpath
-from commoncode import compat
from commoncode import filetype
from commoncode import fileutils
from commoncode.testcase import FileBasedTesting
from commoncode.system import on_windows
-from commoncode.system import py2
-from commoncode.system import py3
"""
Shared archiving test utils.
@@ -50,14 +40,10 @@ def check_size(expected_size, location):
def check_results_with_expected_json(results, expected_loc, regen=False):
if regen:
- if py2:
- wmode = 'wb'
- if py3:
- wmode = 'w'
- with open(expected_loc, wmode) as ex:
+ with open(expected_loc, 'w') as ex:
json.dump(results, ex, indent=2, separators=(',', ':'))
- with open(expected_loc, 'rb') as ex:
- expected = json.load(ex, encoding='utf-8', object_pairs_hook=OrderedDict)
+ with open(expected_loc) as ex:
+ expected = json.load(ex)
try:
assert expected == results
except AssertionError:
@@ -92,13 +78,12 @@ def check_files(test_dir, expected, regen=False):
expected_is_json_file = True
# this is a path to a JSON file
if regen:
- wmode = 'wb' if py2 else 'w'
- with open(expected, wmode) as ex:
+ with open(expected, 'w') as ex:
json.dump(result, ex, indent=2, separators=(',', ':'))
expected_content = result
else:
- with open(expected, 'rb') as ex:
- expected_content = json.load(ex, encoding='utf-8', object_pairs_hook=OrderedDict)
+ with open(expected) as ex:
+ expected_content = json.load(ex)
else:
expected_content = expected
@@ -160,7 +145,7 @@ def to_posix(path):
the windows explorer (except as a UNC or share name). It will be a valid path
everywhere in Python. It will not be valid for windows command line operations.
"""
- is_unicode = isinstance(path, compat.unicode)
+ is_unicode = isinstance(path, str)
ntpath_sep = is_unicode and u'\\' or '\\'
posixpath_sep = is_unicode and u'/' or '/'
if is_posixpath(path):
diff --git a/tests/extractcode/test_archive.py b/tests/extractcode/test_archive.py
index 87c0243..48f2a04 100644
--- a/tests/extractcode/test_archive.py
+++ b/tests/extractcode/test_archive.py
@@ -1,46 +1,34 @@
# -*- coding: utf-8 -*-
#
-# Copyright (c) nexB Inc. and others. All rights reserved.
-# http://nexb.com and https://github.com/nexB/scancode-toolkit/
-# The ScanCode software is licensed under the Apache License version 2.0.
-# Data generated with ScanCode require an acknowledgment.
+# Copyright (c) nexB Inc. and others.
+# SPDX-License-Identifier: Apache-2.0
+#
+# Visit https://aboutcode.org and https://github.com/nexB/ for support and download.
# ScanCode is a trademark of nexB Inc.
#
-# You may not use this software except in compliance with the License.
-# You may obtain a copy of the License at: http://apache.org/licenses/LICENSE-2.0
-# Unless required by applicable law or agreed to in writing, software distributed
-# under the License is distributed on an "AS IS" BASIS, WITHOUT WARRANTIES OR
-# CONDITIONS OF ANY KIND, either express or implied. See the License for the
-# specific language governing permissions and limitations under the License.
+# Licensed under the Apache License, Version 2.0 (the "License");
+# you may not use this file except in compliance with the License.
+# You may obtain a copy of the License at
+#
+# http://www.apache.org/licenses/LICENSE-2.0
#
-# When you publish or redistribute any data created with ScanCode or any ScanCode
-# derivative work, you must accompany this data with the following acknowledgment:
+# Unless required by applicable law or agreed to in writing, software
+# distributed under the License is distributed on an "AS IS" BASIS,
+# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+# See the License for the specific language governing permissions and
+# limitations under the License.
#
-# Generated with ScanCode and provided on an "AS IS" BASIS, WITHOUT WARRANTIES
-# OR CONDITIONS OF ANY KIND, either express or implied. No content created from
-# ScanCode should be considered or used as legal advice. Consult an Attorney
-# for any legal advice.
-# ScanCode is a free software code scanning tool from nexB Inc. and others.
-# Visit https://github.com/nexB/scancode-toolkit/ for support and download.
-
-from __future__ import absolute_import
-from __future__ import print_function
-from __future__ import unicode_literals
-
-import io
+
import os
-from unittest.case import skipIf
import pytest
-import commoncode.date
-from commoncode import compat
+from commoncode import date as commoncode_date
from commoncode import fileutils
from commoncode.system import on_linux
from commoncode.system import on_mac
from commoncode.system import on_windows
-from commoncode.system import py2
-from commoncode.system import py3
+from commoncode.testcase import is_same
from extractcode_assert_utils import BaseArchiveTestCase
from extractcode_assert_utils import check_files
@@ -52,7 +40,6 @@
from extractcode import ExtractErrorFailedToExtract
from extractcode import libarchive2
from extractcode import sevenzip
-from extractcode.libarchive2 import ArchiveError
"""
For each archive type --when possible-- we are testing extraction of:
@@ -395,7 +382,7 @@ def test_extract_targz_with_trailing_data2(self):
test_dir2 = self.get_temp_dir()
test_file2 = self.get_test_loc('archive/tgz/no_trailing.tar.gz')
archive.extract_tar(test_file2, test_dir2)
- assert commoncode.testcase.is_same(test_dir1, test_dir2)
+ assert is_same(test_dir1, test_dir2)
def test_extract_targz_with_mixed_case_and_symlink(self):
test_file = self.get_test_loc('archive/tgz/mixed_case_and_symlink.tgz')
@@ -404,7 +391,7 @@ def test_extract_targz_with_mixed_case_and_symlink(self):
assert [] == result
import json
exp_file = self.get_test_loc('archive/tgz/mixed_case_and_symlink.tgz.expected')
- with io.open(exp_file, encoding='utf-8') as ef:
+ with open(exp_file) as ef:
expected_files = json.load(ef)
check_files(test_dir, list(map(str, expected_files)))
@@ -468,18 +455,8 @@ def test_uncompress_concatenated_gzip(self):
assert b'f1content\nf2content\n' == open(result, 'rb').read()
assert [] == warnings
- @pytest.mark.skipif(py3, reason='Fails for now on Python 3')
- def test_uncompress_gzip_with_trailing_data_py2(self):
- test_file = self.get_test_loc('archive/gzip/trailing_data.gz')
- test_dir = self.get_temp_dir()
- warnings = archive.uncompress_gzip(test_file, test_dir)
- result = os.path.join(test_dir, 'trailing_data.gz-extract')
- assert os.path.exists(result)
- assert [] == warnings
-
- @pytest.mark.xfail
- @pytest.mark.skipif(py2, reason='Fails for now on Python 3')
- def test_uncompress_gzip_with_trailing_data_py3(self):
+ @pytest.mark.xfail(reason='Fails for now on Python 3')
+ def test_uncompress_gzip_with_trailing_data(self):
test_file = self.get_test_loc('archive/gzip/trailing_data.gz')
test_dir = self.get_temp_dir()
warnings = archive.uncompress_gzip(test_file, test_dir)
@@ -619,13 +596,8 @@ def test_uncompress_bzip2_with_trailing_data(self):
def test_uncompress_bzip2_broken(self):
test_file = self.get_test_loc('archive/bz2/bz2_not_tarred_broken.bz2')
test_dir = self.get_temp_dir()
- if py2:
- expected = Exception('invalid data stream')
- else:
- expected = Exception('Invalid data stream')
-
- self.assertRaisesInstance(expected, archive.uncompress_bzip2,
- test_file, test_dir)
+ expected = Exception('Invalid data stream')
+ self.assertRaisesInstance(expected, archive.uncompress_bzip2, test_file, test_dir)
def test_uncompress_bzip2_with_invalid_path(self):
test_file = self.get_test_loc('archive/bz2/bz_invalidpath.bz2')
@@ -650,7 +622,7 @@ def test_sevenzip_extract_can_handle_bz2_multistream_differently(self):
test_dir = self.get_temp_dir()
sevenzip.extract(test_file, test_dir)
expected = self.get_test_loc('archive/bz2/bzip2_multistream/expected.csv')
- # the extraction dir is not created with suffix by z7
+ # the extraction dir is not created with suffix by 7z
result = os.path.join(test_dir, 'example-file.csv')
expected_extracted = open(expected, 'rb').read()
expected_result = open(result, 'rb').read()
@@ -859,19 +831,7 @@ def test_extract_zip_with_relative_path_deeply_nested(self):
except:
assert self.expected_deeply_nested_relative_path_alternative == result
- @pytest.mark.xfail
- @pytest.mark.skipif(on_windows or py3, reason='Expectations are different on Windows')
- def test_extract_zip_with_relative_path_deeply_nested_with_7zip_posix_py2(self):
- test_file = self.get_test_loc('archive/zip/relative_nested.zip')
- test_dir = self.get_temp_dir()
- try:
- sevenzip.extract(test_file, test_dir)
- self.fail('Should raise an exception')
- except ExtractErrorFailedToExtract as e:
- assert 'Unknown extraction error' == str(e)
-
- @pytest.mark.xfail
- @pytest.mark.skipif(on_windows or py2, reason='Expectations are different on Windows')
+ @pytest.mark.xfail(reason='Expectations are different on Windows and this may fail on Windows')
def test_extract_zip_with_relative_path_deeply_nested_with_7zip_posix_py3(self):
test_file = self.get_test_loc('archive/zip/relative_nested.zip')
test_dir = self.get_temp_dir()
@@ -999,7 +959,7 @@ def test_extract_zip_with_timezone(self):
]
# DST sends a monkey wrench.... so we only test the date, not the time
for loc, expected_date in expected:
- result = commoncode.date.get_file_mtime(loc)
+ result = commoncode_date.get_file_mtime(loc)
assert result.startswith(expected_date)
def test_extract_zip_with_timezone_2(self):
@@ -1013,7 +973,7 @@ def test_extract_zip_with_timezone_2(self):
(os.path.join(test_dir, 'primes2.txt'), ('2009-12-05', '2009-12-06',))
]
for loc, expected_date in expected:
- result = commoncode.date.get_file_mtime(loc)
+ result = commoncode_date.get_file_mtime(loc)
assert result.startswith(expected_date)
def test_extract_zip_with_backslash_in_path_1(self):
@@ -1199,12 +1159,7 @@ def test_extract_python_testtar_tar_archive_with_special_files(self):
# https://hg.python.org/cpython/raw-file/bff88c866886/Lib/test/testtar.tar
test_dir = self.get_temp_dir()
result = archive.extract_tar(test_file, test_dir)
- if py2:
- expected_warnings = [
- "'pax/bad-pax-\\xe4\\xf6\\xfc': \nPathname can't be converted from UTF-8 to current locale."]
- else:
- expected_warnings = [
- u"'pax/bad-pax-äöü': \nPathname can't be converted from UTF-8 to current locale."]
+ expected_warnings = [u"'pax/bad-pax-äöü': \nPathname can't be converted from UTF-8 to current locale."]
assert sorted(expected_warnings) == sorted(result)
@@ -1240,8 +1195,6 @@ def test_extract_python_testtar_tar_archive_with_special_files(self):
'ustar/sparse',
'ustar/umlauts-AOUaouss'
]
- if on_linux and py2:
- expected = [bytes(e) for e in expected]
check_files(test_dir, expected)
def test_extract_rubygem(self):
@@ -1249,8 +1202,6 @@ def test_extract_rubygem(self):
test_dir = self.get_temp_dir()
archive.extract_tar(test_file, test_dir)
expected = ['checksums.yaml.gz', 'data.tar.gz', 'metadata.gz']
- if on_linux and py2:
- expected = [bytes(e) for e in expected]
check_files(test_dir, expected)
@@ -1316,7 +1267,7 @@ def test_extract_ar_verify_dates(self):
]
# DST sends a monkey wrench.... so we only test the date, not the time
for loc, expected_date in expected:
- result = commoncode.date.get_file_mtime(loc)
+ result = commoncode_date.get_file_mtime(loc)
assert result.startswith(expected_date)
def test_extract_ar_broken_7z(self):
@@ -1384,10 +1335,6 @@ def test_extract_ar_with_relative_path_and_backslashes_in_names_libarch(self):
# 7zip is better, but has a security bug for now
# GNU ar works fine otherwise, but there are portability issues
expected = ['dot', 'dot_1']
-
- if on_linux and py2:
- expected = [bytes(e) for e in expected]
-
check_files(test_dir, expected)
def test_extract_ar_with_relative_path_and_backslashes_in_names_7z(self):
@@ -1515,9 +1462,6 @@ def test_extract_cpio_broken2(self):
test_dir = self.get_temp_dir()
result = archive.extract_cpio(test_file, test_dir)
expected = sorted(['elfinfo-1.0.tar.gz', 'elfinfo.spec'])
- if on_linux and py2:
- expected = [e.encode('utf-8') for e in expected]
-
assert expected == sorted(os.listdir(test_dir))
assert ["'elfinfo.spec': \nSkipped 72 bytes before finding valid header"] == result
@@ -1785,13 +1729,12 @@ def test_extract_rar_with_password(self):
test_file = self.get_test_loc('archive/rar/rar_password.rar')
test_dir = self.get_temp_dir()
expected = Exception('Prefix found')
- self.assertRaisesInstance(expected, archive.extract_rar,
- test_file, test_dir)
+ self.assertRaisesInstance(expected, archive.extract_rar, test_file, test_dir)
def test_extract_rar_with_non_ascii_path(self):
test_file = self.get_test_loc('archive/rar/non_ascii_corrupted.rar')
# The bug only occurs if the path was given as Unicode
- test_file = compat.unicode(test_file)
+ test_file = str(test_file)
test_dir = self.get_temp_dir()
# raise an exception but still extracts some
expected = Exception('Prefix found')
@@ -1999,16 +1942,15 @@ def test_extract_dia_basic(self):
result = os.path.join(test_dir, 'dia.dia-extract')
assert os.path.exists(result)
- @pytest.mark.skipif(py3, reason='Fails for now on Python 3')
- def test_extract_dia_with_trailing_data_py2(self):
+ @pytest.mark.xfail(reason='Fails for now on Python 3')
+ def test_extract_dia_with_trailing_data(self):
test_file = self.get_test_loc('archive/dia/dia_trailing.dia')
test_dir = self.get_temp_dir()
archive.uncompress_gzip(test_file, test_dir)
result = os.path.join(test_dir, 'dia_trailing.dia-extract')
assert os.path.exists(result)
- @pytest.mark.xfail
- @pytest.mark.skipif(py2, reason='Fails for now on Python 3')
+ @pytest.mark.xfail(reason='Fails for now on Python 3')
def test_extract_dia_with_trailing_data_py3(self):
test_file = self.get_test_loc('archive/dia/dia_trailing.dia')
test_dir = self.get_temp_dir()
@@ -2278,8 +2220,8 @@ def check_extract_weird_names(
listed in the `test_file.excepted` file exist in the extracted target
directory. Regen expected file if True.
"""
- if not isinstance(test_file, compat.unicode):
- test_file = compat.unicode(test_file)
+ if not isinstance(test_file, str):
+ test_file = str(test_file)
test_file = self.get_test_loc(test_file)
test_dir = self.get_temp_dir()
@@ -2299,7 +2241,7 @@ def check_extract_weird_names(
len_test_dir = len(test_dir)
extracted = sorted(path[len_test_dir:] for path in fileutils.resource_iter(test_dir, with_dirs=False))
- extracted = [compat.unicode(p) for p in extracted]
+ extracted = [str(p) for p in extracted]
extracted = [to_posix(p) for p in extracted]
if on_linux:
@@ -2312,11 +2254,7 @@ def check_extract_weird_names(
expected_file = test_file + '_' + expected_suffix + '_' + os_suffix + '.expected'
import json
if regen:
- if py2:
- wmode = 'wb'
- if py3:
- wmode = 'w'
- with open(expected_file, wmode) as ef:
+ with open(expected_file, 'w') as ef:
ef.write(json.dumps(extracted, indent=2))
expected = json.loads(open(expected_file).read())
@@ -2379,7 +2317,7 @@ def test_extract_ar_with_weird_filenames_with_libarchive_win(self):
self.check_extract_weird_names(
libarchive2.extract, test_file, expected_warnings=[], expected_suffix='libarch')
self.fail('Exception not raised.')
- except ArchiveError as ae:
+ except libarchive2.ArchiveError as ae:
assert str(ae).startswith('Incorrect file header signature')
def test_extract_cpio_with_weird_filenames_with_libarchive_win(self):
@@ -2677,57 +2615,23 @@ def test_extract_zip_with_weird_filenames_with_sevenzip_win(self):
@pytest.mark.skipif(not on_windows, reason='Run only on Windows because of specific test expectations.')
class TestExtractArchiveWithIllegalFilenamesWithSevenzipOnWinWarning(ExtractArchiveWithIllegalFilenamesTestCase):
- if py2:
-
- # The results are not correct but not a problem: we use libarchive for these
- @pytest.mark.xfail
- def test_extract_7zip_with_weird_filenames_with_sevenzip_win(self):
- test_file = self.get_test_loc('archive/weird_names/weird_names.7z')
- self.check_extract_weird_names(
- sevenzip.extract, test_file, expected_warnings=[], expected_suffix='7zip',
- check_warnings=True, check_only_warnings=True)
-
- else:
-
- def test_extract_7zip_with_weird_filenames_with_sevenzip_win(self):
- test_file = self.get_test_loc('archive/weird_names/weird_names.7z')
- self.check_extract_weird_names(
- sevenzip.extract, test_file, expected_warnings=[], expected_suffix='7zip',
- check_warnings=True, check_only_warnings=True)
-
- if py2:
-
- @pytest.mark.xfail # not a problem: we use libarchive for these
- def test_extract_ar_with_weird_filenames_with_sevenzip_win(self):
- test_file = self.get_test_loc('archive/weird_names/weird_names.ar')
- self.check_extract_weird_names(
- sevenzip.extract, test_file, expected_warnings=[], expected_suffix='7zip',
- check_warnings=True, check_only_warnings=True)
-
- else:
-
- def test_extract_ar_with_weird_filenames_with_sevenzip_win(self):
- test_file = self.get_test_loc('archive/weird_names/weird_names.ar')
- self.check_extract_weird_names(
- sevenzip.extract, test_file, expected_warnings=[], expected_suffix='7zip',
- check_warnings=True, check_only_warnings=True)
-
- if py2:
-
- @pytest.mark.xfail # not a problem: we use libarchive for these
- def test_extract_cpio_with_weird_filenames_with_sevenzip_win(self):
- test_file = self.get_test_loc('archive/weird_names/weird_names.cpio')
- self.check_extract_weird_names(
- sevenzip.extract, test_file, expected_warnings=[], expected_suffix='7zip',
- check_warnings=True, check_only_warnings=True)
+ def test_extract_7zip_with_weird_filenames_with_sevenzip_win(self):
+ test_file = self.get_test_loc('archive/weird_names/weird_names.7z')
+ self.check_extract_weird_names(
+ sevenzip.extract, test_file, expected_warnings=[], expected_suffix='7zip',
+ check_warnings=True, check_only_warnings=True)
- else:
+ def test_extract_ar_with_weird_filenames_with_sevenzip_win(self):
+ test_file = self.get_test_loc('archive/weird_names/weird_names.ar')
+ self.check_extract_weird_names(
+ sevenzip.extract, test_file, expected_warnings=[], expected_suffix='7zip',
+ check_warnings=True, check_only_warnings=True)
- def test_extract_cpio_with_weird_filenames_with_sevenzip_win(self):
- test_file = self.get_test_loc('archive/weird_names/weird_names.cpio')
- self.check_extract_weird_names(
- sevenzip.extract, test_file, expected_warnings=[], expected_suffix='7zip',
- check_warnings=True, check_only_warnings=True)
+ def test_extract_cpio_with_weird_filenames_with_sevenzip_win(self):
+ test_file = self.get_test_loc('archive/weird_names/weird_names.cpio')
+ self.check_extract_weird_names(
+ sevenzip.extract, test_file, expected_warnings=[], expected_suffix='7zip',
+ check_warnings=True, check_only_warnings=True)
def test_extract_iso_with_weird_filenames_with_sevenzip_win(self):
test_file = self.get_test_loc('archive/weird_names/weird_names.iso')
@@ -2748,22 +2652,11 @@ def test_extract_tar_with_weird_filenames_with_sevenzip_win(self):
sevenzip.extract, test_file, expected_warnings=[], expected_suffix='7zip',
check_warnings=True, check_only_warnings=True)
- if py2:
-
- @pytest.mark.xfail # not a problem: we use libarchive for these
- def test_extract_zip_with_weird_filenames_with_sevenzip_win(self):
- test_file = self.get_test_loc('archive/weird_names/weird_names.zip')
- self.check_extract_weird_names(
- sevenzip.extract, test_file, expected_warnings=[], expected_suffix='7zip',
- check_warnings=True, check_only_warnings=True)
-
- else:
-
- def test_extract_zip_with_weird_filenames_with_sevenzip_win(self):
- test_file = self.get_test_loc('archive/weird_names/weird_names.zip')
- self.check_extract_weird_names(
- sevenzip.extract, test_file, expected_warnings=[], expected_suffix='7zip',
- check_warnings=True, check_only_warnings=True)
+ def test_extract_zip_with_weird_filenames_with_sevenzip_win(self):
+ test_file = self.get_test_loc('archive/weird_names/weird_names.zip')
+ self.check_extract_weird_names(
+ sevenzip.extract, test_file, expected_warnings=[], expected_suffix='7zip',
+ check_warnings=True, check_only_warnings=True)
class TestZipSlip(BaseArchiveTestCase):
diff --git a/tests/extractcode/test_extract.py b/tests/extractcode/test_extract.py
index 6629710..b6f89e9 100644
--- a/tests/extractcode/test_extract.py
+++ b/tests/extractcode/test_extract.py
@@ -1,29 +1,23 @@
# -*- coding: utf-8 -*-
#
-# Copyright (c) 2015 nexB Inc. and others. All rights reserved.
-# http://nexb.com and https://github.com/nexB/scancode-toolkit/
-# The ScanCode software is licensed under the Apache License version 2.0.
-# Data generated with ScanCode require an acknowledgment.
+# Copyright (c) nexB Inc. and others.
+# SPDX-License-Identifier: Apache-2.0
+#
+# Visit https://aboutcode.org and https://github.com/nexB/ for support and download.
# ScanCode is a trademark of nexB Inc.
#
-# You may not use this software except in compliance with the License.
-# You may obtain a copy of the License at: http://apache.org/licenses/LICENSE-2.0
-# Unless required by applicable law or agreed to in writing, software distributed
-# under the License is distributed on an "AS IS" BASIS, WITHOUT WARRANTIES OR
-# CONDITIONS OF ANY KIND, either express or implied. See the License for the
-# specific language governing permissions and limitations under the License.
+# Licensed under the Apache License, Version 2.0 (the "License");
+# you may not use this file except in compliance with the License.
+# You may obtain a copy of the License at
#
-# When you publish or redistribute any data created with ScanCode or any ScanCode
-# derivative work, you must accompany this data with the following acknowledgment:
+# http://www.apache.org/licenses/LICENSE-2.0
+#
+# Unless required by applicable law or agreed to in writing, software
+# distributed under the License is distributed on an "AS IS" BASIS,
+# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+# See the License for the specific language governing permissions and
+# limitations under the License.
#
-# Generated with ScanCode and provided on an "AS IS" BASIS, WITHOUT WARRANTIES
-# OR CONDITIONS OF ANY KIND, either express or implied. No content created from
-# ScanCode should be considered or used as legal advice. Consult an Attorney
-# for any legal advice.
-# ScanCode is a free software code scanning tool from nexB Inc. and others.
-# Visit https://github.com/nexB/scancode-toolkit/ for support and download.
-
-from __future__ import absolute_import, print_function
import io
import os
@@ -35,7 +29,6 @@
from commoncode.fileutils import as_posixpath
from commoncode.system import on_linux
from commoncode.system import on_windows
-from commoncode.system import py3
from commoncode.testcase import FileBasedTesting
import extractcode
@@ -386,7 +379,6 @@ def test_extract_tree_shallow_then_recursive(self):
def test_uncompress_corrupted_archive_with_zlib(self):
from extractcode import archive
- import zlib
test_file = self.get_test_loc('extract/corrupted/a.tar.gz', copy=True)
test_dir = self.get_temp_dir()
expected = Exception('Error -3 while decompressing')
@@ -399,7 +391,7 @@ def test_uncompress_corrupted_archive_with_libarchive(self):
expected = Exception('gzip decompression failed')
self.assertRaisesInstance(expected, libarchive2.extract, test_file, test_dir)
- @pytest.mark.skipif(py3 and not on_linux, reason='Expectations are different on Windows and macOS')
+ @pytest.mark.skipif(not on_linux, reason='Expectations are different on Windows and macOS')
def test_extract_tree_with_corrupted_archives_linux(self):
expected = (
'a.tar.gz',
@@ -413,7 +405,7 @@ def test_extract_tree_with_corrupted_archives_linux(self):
assert result.errors[0].startswith('gzip decompression failed')
assert not result.warnings
- @pytest.mark.skipif(py3 and on_linux, reason='Expectations are different on Windows and macOS')
+ @pytest.mark.skipif(on_linux, reason='Expectations are different on Windows and macOS')
def test_extract_tree_with_corrupted_archives_mac_win(self):
expected = (
'a.tar.gz',
@@ -857,7 +849,7 @@ def test_extract_always_returns_a_generator_and_not_a_list(self):
test_dir = self.get_test_loc('extract/generator', copy=True)
result = extract.extract(test_dir)
assert isinstance(result, GeneratorType)
-
+
def test_extract_ignore_file(self):
test_dir = self.get_test_loc('extract/ignore', copy=True)
expected = [
@@ -869,7 +861,6 @@ def test_extract_ignore_file(self):
'gamma/gamma.zip',
'gamma/gamma.zip-extract/c.txt'
]
- from extractcode import default_kinds
result = list(extract.extract(test_dir, recurse=True, ignore_pattern=('alpha.zip',)))
check_no_error(result)
check_files(test_dir, expected)
@@ -889,7 +880,6 @@ def test_extract_ignore_directory(self):
'beta.tar-extract/c.txt',
'gamma/gamma.zip',
]
- from extractcode import default_kinds
result = list(extract.extract(test_dir, recurse=True, ignore_pattern=('gamma',)))
check_no_error(result)
check_files(test_dir, expected)
@@ -909,7 +899,6 @@ def test_extract_ignore_pattern(self):
'gamma/gamma.zip',
'gamma/gamma.zip-extract/c.txt'
]
- from extractcode import default_kinds
result = list(extract.extract(test_dir, recurse=True, ignore_pattern=('b*.zip',)))
check_no_error(result)
- check_files(test_dir, expected)
\ No newline at end of file
+ check_files(test_dir, expected)
diff --git a/tests/extractcode/test_extractcode.py b/tests/extractcode/test_extractcode.py
index 3ed0a7d..0a7dafb 100644
--- a/tests/extractcode/test_extractcode.py
+++ b/tests/extractcode/test_extractcode.py
@@ -1,29 +1,23 @@
+
+#
+# Copyright (c) nexB Inc. and others.
+# SPDX-License-Identifier: Apache-2.0
#
-# Copyright (c) 2017 nexB Inc. and others. All rights reserved.
-# http://nexb.com and https://github.com/nexB/scancode-toolkit/
-# The ScanCode software is licensed under the Apache License version 2.0.
-# Data generated with ScanCode require an acknowledgment.
+# Visit https://aboutcode.org and https://github.com/nexB/ for support and download.
# ScanCode is a trademark of nexB Inc.
#
-# You may not use this software except in compliance with the License.
-# You may obtain a copy of the License at: http://apache.org/licenses/LICENSE-2.0
-# Unless required by applicable law or agreed to in writing, software distributed
-# under the License is distributed on an "AS IS" BASIS, WITHOUT WARRANTIES OR
-# CONDITIONS OF ANY KIND, either express or implied. See the License for the
-# specific language governing permissions and limitations under the License.
+# Licensed under the Apache License, Version 2.0 (the "License");
+# you may not use this file except in compliance with the License.
+# You may obtain a copy of the License at
+#
+# http://www.apache.org/licenses/LICENSE-2.0
#
-# When you publish or redistribute any data created with ScanCode or any ScanCode
-# derivative work, you must accompany this data with the following acknowledgment:
+# Unless required by applicable law or agreed to in writing, software
+# distributed under the License is distributed on an "AS IS" BASIS,
+# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+# See the License for the specific language governing permissions and
+# limitations under the License.
#
-# Generated with ScanCode and provided on an "AS IS" BASIS, WITHOUT WARRANTIES
-# OR CONDITIONS OF ANY KIND, either express or implied. No content created from
-# ScanCode should be considered or used as legal advice. Consult an Attorney
-# for any legal advice.
-# ScanCode is a free software code scanning tool from nexB Inc. and others.
-# Visit https://github.com/nexB/scancode-toolkit/ for support and download.
-
-from __future__ import absolute_import
-from __future__ import print_function
from os.path import dirname
from os.path import exists
diff --git a/tests/extractcode/test_extractcode_cli.py b/tests/extractcode/test_extractcode_cli.py
index 19e621a..bad30c9 100644
--- a/tests/extractcode/test_extractcode_cli.py
+++ b/tests/extractcode/test_extractcode_cli.py
@@ -1,31 +1,23 @@
+
+#
+# Copyright (c) nexB Inc. and others.
+# SPDX-License-Identifier: Apache-2.0
#
-# Copyright (c) 2017 nexB Inc. and others. All rights reserved.
-# http://nexb.com and https://github.com/nexB/scancode-toolkit/
-# The ScanCode software is licensed under the Apache License version 2.0.
-# Data generated with ScanCode require an acknowledgment.
+# Visit https://aboutcode.org and https://github.com/nexB/ for support and download.
# ScanCode is a trademark of nexB Inc.
#
-# You may not use this software except in compliance with the License.
-# You may obtain a copy of the License at: http://apache.org/licenses/LICENSE-2.0
-# Unless required by applicable law or agreed to in writing, software distributed
-# under the License is distributed on an "AS IS" BASIS, WITHOUT WARRANTIES OR
-# CONDITIONS OF ANY KIND, either express or implied. See the License for the
-# specific language governing permissions and limitations under the License.
+# Licensed under the Apache License, Version 2.0 (the "License");
+# you may not use this file except in compliance with the License.
+# You may obtain a copy of the License at
+#
+# http://www.apache.org/licenses/LICENSE-2.0
#
-# When you publish or redistribute any data created with ScanCode or any ScanCode
-# derivative work, you must accompany this data with the following acknowledgment:
+# Unless required by applicable law or agreed to in writing, software
+# distributed under the License is distributed on an "AS IS" BASIS,
+# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+# See the License for the specific language governing permissions and
+# limitations under the License.
#
-# Generated with ScanCode and provided on an "AS IS" BASIS, WITHOUT WARRANTIES
-# OR CONDITIONS OF ANY KIND, either express or implied. No content created from
-# ScanCode should be considered or used as legal advice. Consult an Attorney
-# for any legal advice.
-# ScanCode is a free software code scanning tool from nexB Inc. and others.
-# Visit https://github.com/nexB/scancode-toolkit/ for support and download.
-
-from __future__ import absolute_import
-from __future__ import print_function
-from __future__ import division
-from __future__ import unicode_literals
import os
@@ -34,14 +26,12 @@
import pytest
from commoncode.fileutils import as_posixpath
-from commoncode.fileutils import fsencode
from commoncode.fileutils import resource_iter
from commoncode.testcase import FileDrivenTesting
from commoncode.system import on_linux
from commoncode.system import on_windows
-from commoncode.system import py2
-from extractcode import cli
+from extractcode import cli
test_env = FileDrivenTesting()
test_env.test_data_dir = os.path.join(os.path.dirname(__file__), 'data')
@@ -51,8 +41,6 @@
the actual command outputs as if using a TTY or not.
"""
-EMPTY_STRING = b'' if on_linux and py2 else ''
-
def test_extractcode_command_can_take_an_empty_directory(monkeypatch):
test_dir = test_env.get_temp_dir()
@@ -186,19 +174,16 @@ def test_usage_and_help_return_a_correct_script_name_on_all_platforms(monkeypatc
def test_extractcode_command_can_extract_archive_with_unicode_names_verbose(monkeypatch):
monkeypatch.setattr(click._termui_impl, 'isatty', lambda _: True)
test_dir = test_env.get_test_loc('cli/unicodearch', copy=True)
- if on_linux and py2:
- test_dir = fsencode(test_dir)
runner = CliRunner()
result = runner.invoke(cli.extractcode, ['--verbose', test_dir])
assert result.exit_code == 0
assert 'Sanders' in result.output
- uni_arch = b'unicodepath.tgz' if on_linux and py2 else u'unicodepath.tgz'
- uni_path = b'/unicodepath/' if on_linux and py2 else u'/unicodepath/'
-
- file_result = [f for f in map(as_posixpath, resource_iter(test_dir, with_dirs=False)) if not f.endswith(uni_arch)]
- file_result = [EMPTY_STRING.join(f.partition(uni_path)[1:]) for f in file_result]
+ file_result = [
+ f for f in map(as_posixpath, resource_iter(test_dir, with_dirs=False))
+ if not f.endswith('unicodepath.tgz')]
+ file_result = [''.join(f.partition('/unicodepath/')[1:]) for f in file_result]
file_result = [f for f in file_result if f]
expected = [
'/unicodepath/Ho_',
@@ -211,17 +196,14 @@ def test_extractcode_command_can_extract_archive_with_unicode_names_verbose(monk
def test_extractcode_command_can_extract_archive_with_unicode_names(monkeypatch):
monkeypatch.setattr(click._termui_impl, 'isatty', lambda _: True)
test_dir = test_env.get_test_loc('cli/unicodearch', copy=True)
- if on_linux:
- test_dir = fsencode(test_dir)
runner = CliRunner()
result = runner.invoke(cli.extractcode, [test_dir])
assert result.exit_code == 0
- uni_arch = b'unicodepath.tgz' if on_linux and py2 else 'unicodepath.tgz'
- uni_path = b'/unicodepath/' if on_linux and py2 else '/unicodepath/'
-
- file_result = [f for f in map(as_posixpath, resource_iter(test_dir, with_dirs=False)) if not f.endswith(uni_arch)]
- file_result = [EMPTY_STRING.join(f.partition(uni_path)[1:]) for f in file_result]
+ file_result = [
+ f for f in map(as_posixpath, resource_iter(test_dir, with_dirs=False))
+ if not f.endswith('unicodepath.tgz')]
+ file_result = [''.join(f.partition('/unicodepath/')[1:]) for f in file_result]
file_result = [f for f in file_result if f]
expected = [
'/unicodepath/Ho_',
@@ -237,7 +219,9 @@ def test_extractcode_command_can_extract_shallow(monkeypatch):
runner = CliRunner()
result = runner.invoke(cli.extractcode, ['--shallow', test_dir])
assert result.exit_code == 0
- file_result = [f for f in map(as_posixpath, resource_iter(test_dir, with_dirs=False)) if not f.endswith('unicodepath.tgz')]
+ file_result = [
+ f for f in map(as_posixpath, resource_iter(test_dir, with_dirs=False))
+ if not f.endswith('unicodepath.tgz')]
file_result = [''.join(f.partition('/top.zip-extract/')[1:]) for f in file_result]
file_result = [f for f in file_result if f]
# this checks that the zip in top.zip are not extracted
@@ -248,17 +232,18 @@ def test_extractcode_command_can_extract_shallow(monkeypatch):
]
assert sorted(expected) == sorted(file_result)
+
def test_extractcode_command_can_ignore(monkeypatch):
monkeypatch.setattr(click._termui_impl, 'isatty', lambda _: True)
test_dir = test_env.get_test_loc('cli/extract_ignore', copy=True)
- if on_linux:
- test_dir = fsencode(test_dir)
runner = CliRunner()
result = runner.invoke(cli.extractcode, ['--ignore', '*.tar', test_dir])
assert result.exit_code == 0
- file_result = [f for f in map(as_posixpath, resource_iter(test_dir, with_dirs=False)) if not f.endswith('a.tar') or not f.endswith('b.tar')]
- file_result = [EMPTY_STRING.join(f.partition('/a.zip-extract/')[1:]) for f in file_result]
+ file_result = [
+ f for f in map(as_posixpath, resource_iter(test_dir, with_dirs=False))
+ if not f.endswith('a.tar') or not f.endswith('b.tar')]
+ file_result = [''.join(f.partition('/a.zip-extract/')[1:]) for f in file_result]
file_result = [f for f in file_result if f]
expected = [
'/a.zip-extract/a.txt',
@@ -268,6 +253,7 @@ def test_extractcode_command_can_ignore(monkeypatch):
]
assert sorted(expected) == sorted(file_result)
+
@pytest.mark.skipif(on_windows, reason='FIXME: this test fails on Windows until we have support for long file names.')
def test_extractcode_command_can_extract_nuget(monkeypatch):
test_dir = test_env.get_test_loc('cli/extract_nuget', copy=True)
@@ -276,4 +262,4 @@ def test_extractcode_command_can_extract_nuget(monkeypatch):
result = runner.invoke(cli.extractcode, ['--verbose', test_dir], catch_exceptions=False)
if result.exit_code != 0:
print(result.output)
- assert 'ERROR extracting' not in result.output
\ No newline at end of file
+ assert 'ERROR extracting' not in result.output
diff --git a/tests/extractcode/test_libarchive2.py b/tests/extractcode/test_libarchive2.py
index 9980f5f..783ac39 100644
--- a/tests/extractcode/test_libarchive2.py
+++ b/tests/extractcode/test_libarchive2.py
@@ -1,41 +1,34 @@
+
+#
+# Copyright (c) nexB Inc. and others.
+# SPDX-License-Identifier: Apache-2.0
#
-# Copyright (c) 2017 nexB Inc. and others. All rights reserved.
-# http://nexb.com and https://github.com/nexB/scancode-toolkit/
-# The ScanCode software is licensed under the Apache License version 2.0.
-# Data generated with ScanCode require an acknowledgment.
+# Visit https://aboutcode.org and https://github.com/nexB/ for support and download.
# ScanCode is a trademark of nexB Inc.
#
-# You may not use this software except in compliance with the License.
-# You may obtain a copy of the License at: http://apache.org/licenses/LICENSE-2.0
-# Unless required by applicable law or agreed to in writing, software distributed
-# under the License is distributed on an "AS IS" BASIS, WITHOUT WARRANTIES OR
-# CONDITIONS OF ANY KIND, either express or implied. See the License for the
-# specific language governing permissions and limitations under the License.
+# Licensed under the Apache License, Version 2.0 (the "License");
+# you may not use this file except in compliance with the License.
+# You may obtain a copy of the License at
#
-# When you publish or redistribute any data created with ScanCode or any ScanCode
-# derivative work, you must accompany this data with the following acknowledgment:
+# http://www.apache.org/licenses/LICENSE-2.0
+#
+# Unless required by applicable law or agreed to in writing, software
+# distributed under the License is distributed on an "AS IS" BASIS,
+# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+# See the License for the specific language governing permissions and
+# limitations under the License.
#
-# Generated with ScanCode and provided on an "AS IS" BASIS, WITHOUT WARRANTIES
-# OR CONDITIONS OF ANY KIND, either express or implied. No content created from
-# ScanCode should be considered or used as legal advice. Consult an Attorney
-# for any legal advice.
-# ScanCode is a free software code scanning tool from nexB Inc. and others.
-# Visit https://github.com/nexB/scancode-toolkit/ for support and download.
-
-from __future__ import absolute_import
-from __future__ import print_function
-from __future__ import unicode_literals
from commoncode import fileutils
from extractcode_assert_utils import check_files
from extractcode_assert_utils import BaseArchiveTestCase
-
"""
Minimal smoke tests for libarchive2.
"""
+
class TestExtractorTest(BaseArchiveTestCase):
def test_libarchive_extract_can_extract_to_relative_paths(self):
diff --git a/tests/extractcode/test_patch.py b/tests/extractcode/test_patch.py
index d396a54..6ecb3cc 100644
--- a/tests/extractcode/test_patch.py
+++ b/tests/extractcode/test_patch.py
@@ -1,40 +1,32 @@
+
+#
+# Copyright (c) nexB Inc. and others.
+# SPDX-License-Identifier: Apache-2.0
#
-# Copyright (c) 2017 nexB Inc. and others. All rights reserved.
-# http://nexb.com and https://github.com/nexB/scancode-toolkit/
-# The ScanCode software is licensed under the Apache License version 2.0.
-# Data generated with ScanCode require an acknowledgment.
+# Visit https://aboutcode.org and https://github.com/nexB/ for support and download.
# ScanCode is a trademark of nexB Inc.
#
-# You may not use this software except in compliance with the License.
-# You may obtain a copy of the License at: http://apache.org/licenses/LICENSE-2.0
-# Unless required by applicable law or agreed to in writing, software distributed
-# under the License is distributed on an "AS IS" BASIS, WITHOUT WARRANTIES OR
-# CONDITIONS OF ANY KIND, either express or implied. See the License for the
-# specific language governing permissions and limitations under the License.
+# Licensed under the Apache License, Version 2.0 (the "License");
+# you may not use this file except in compliance with the License.
+# You may obtain a copy of the License at
#
-# When you publish or redistribute any data created with ScanCode or any ScanCode
-# derivative work, you must accompany this data with the following acknowledgment:
+# http://www.apache.org/licenses/LICENSE-2.0
+#
+# Unless required by applicable law or agreed to in writing, software
+# distributed under the License is distributed on an "AS IS" BASIS,
+# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+# See the License for the specific language governing permissions and
+# limitations under the License.
#
-# Generated with ScanCode and provided on an "AS IS" BASIS, WITHOUT WARRANTIES
-# OR CONDITIONS OF ANY KIND, either express or implied. No content created from
-# ScanCode should be considered or used as legal advice. Consult an Attorney
-# for any legal advice.
-# ScanCode is a free software code scanning tool from nexB Inc. and others.
-# Visit https://github.com/nexB/scancode-toolkit/ for support and download.
-
-from __future__ import absolute_import
-from __future__ import print_function
-from __future__ import unicode_literals
import io
import json
import os
from unittest.case import expectedFailure
-from commoncode.system import py2
-from commoncode.system import py3
from commoncode.testcase import FileBasedTesting
from commoncode.text import as_unicode
+
from extractcode import patch
@@ -64,14 +56,9 @@ def check_patch(test_file, expected_file, regen=False):
for s, t, lines in result]
if regen:
- if py2:
- wmode = 'wb'
- if py3:
- wmode = 'w'
-
- with io.open(expected_file, wmode) as regened:
+ with io.open(expected_file, 'w') as regened:
json.dump(result, regened, indent=2)
- with io.open(expected_file, encoding='utf-8') as expect:
+ with open(expected_file) as expect:
expected = json.load(expect)
assert expected == result
diff --git a/tests/extractcode/test_sevenzip.py b/tests/extractcode/test_sevenzip.py
index a5dd2bf..ee32439 100644
--- a/tests/extractcode/test_sevenzip.py
+++ b/tests/extractcode/test_sevenzip.py
@@ -1,42 +1,35 @@
#
-# Copyright (c) nexB Inc. and others. All rights reserved.
-# http://nexb.com and https://github.com/nexB/scancode-toolkit/
-# The ScanCode software is licensed under the Apache License version 2.0.
-# Data generated with ScanCode require an acknowledgment.
+# Copyright (c) nexB Inc. and others.
+# SPDX-License-Identifier: Apache-2.0
+#
+# Visit https://aboutcode.org and https://github.com/nexB/ for support and download.
# ScanCode is a trademark of nexB Inc.
#
-# You may not use this software except in compliance with the License.
-# You may obtain a copy of the License at: http://apache.org/licenses/LICENSE-2.0
-# Unless required by applicable law or agreed to in writing, software distributed
-# under the License is distributed on an "AS IS" BASIS, WITHOUT WARRANTIES OR
-# CONDITIONS OF ANY KIND, either express or implied. See the License for the
-# specific language governing permissions and limitations under the License.
+# Licensed under the Apache License, Version 2.0 (the "License");
+# you may not use this file except in compliance with the License.
+# You may obtain a copy of the License at
#
-# When you publish or redistribute any data created with ScanCode or any ScanCode
-# derivative work, you must accompany this data with the following acknowledgment:
+# http://www.apache.org/licenses/LICENSE-2.0
+#
+# Unless required by applicable law or agreed to in writing, software
+# distributed under the License is distributed on an "AS IS" BASIS,
+# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+# See the License for the specific language governing permissions and
+# limitations under the License.
#
-# Generated with ScanCode and provided on an "AS IS" BASIS, WITHOUT WARRANTIES
-# OR CONDITIONS OF ANY KIND, either express or implied. No content created from
-# ScanCode should be considered or used as legal advice. Consult an Attorney
-# for any legal advice.
-# ScanCode is a free software code scanning tool from nexB Inc. and others.
-# Visit https://github.com/nexB/scancode-toolkit/ for support and download.
-
-from __future__ import absolute_import
-from __future__ import print_function
import os
import json
import posixpath
-from unittest.case import skipIf
+import pytest
+
+from commoncode import fileutils
from commoncode.testcase import FileBasedTesting
-from commoncode.system import py2
-from commoncode.system import py3
from commoncode.system import on_windows
-from commoncode import fileutils
-from extractcode import sevenzip
+
from extractcode import ExtractErrorFailedToExtract
+from extractcode import sevenzip
class TestSevenZip(FileBasedTesting):
@@ -44,15 +37,11 @@ class TestSevenZip(FileBasedTesting):
def check_results_with_expected_json(self, results, expected_loc, clean_dates=False, regen=False):
if regen:
- if py2:
- wmode = 'wb'
- if py3:
- wmode = 'w'
- with open(expected_loc, wmode) as ex:
+ with open(expected_loc, 'w') as ex:
json.dump(results, ex, indent=2, separators=(',', ':'))
- with open(expected_loc, 'rb') as ex:
- expected = json.load(ex, encoding='utf-8')
+ with open(expected_loc) as ex:
+ expected = json.load(ex)
if clean_dates:
if isinstance(results, list):
self.clean_dates(results)
@@ -147,7 +136,7 @@ def test_extract_of_tar_with_aboslute_path(self):
class TestSevenZipListEntries(TestSevenZip):
- @skipIf(on_windows, 'Windows file-by-file extracton is not working well')
+ @pytest.mark.skipif(on_windows, reason='Windows file-by-file extracton is not working well')
def test_list_entries_of_special_tar(self):
test_loc = self.get_test_loc('sevenzip/special.tar')
expected_loc = test_loc + '-entries-expected.json'
@@ -157,7 +146,7 @@ def test_list_entries_of_special_tar(self):
results = entries + errors
self.check_results_with_expected_json(results, expected_loc, regen=False)
- @skipIf(not on_windows, 'Windows file-by-file extracton is not working well')
+ @pytest.mark.skipif(not on_windows, reason='Windows file-by-file extracton is not working well')
def test_list_entries_of_special_tar_win(self):
test_loc = self.get_test_loc('sevenzip/special.tar')
expected_loc = test_loc + '-entries-expected-win.json'
@@ -167,7 +156,7 @@ def test_list_entries_of_special_tar_win(self):
results = entries + errors
self.check_results_with_expected_json(results, expected_loc, clean_dates=True, regen=False)
- @skipIf(on_windows, 'Windows file-by-file extracton is not working well')
+ @pytest.mark.skipif(on_windows, reason='Windows file-by-file extracton is not working well')
def test_list_entries_with_weird_names_7z(self):
test_loc = self.get_test_loc('sevenzip/weird_names.7z')
expected_loc = test_loc + '-entries-expected.json'
@@ -177,7 +166,7 @@ def test_list_entries_with_weird_names_7z(self):
results = entries + errors
self.check_results_with_expected_json(results, expected_loc, regen=False)
- @skipIf(not on_windows, 'Windows file-by-file extracton is not working well')
+ @pytest.mark.skipif(not on_windows, reason='Windows file-by-file extracton is not working well')
def test_list_entries_with_weird_names_7z_win(self):
test_loc = self.get_test_loc('sevenzip/weird_names.7z')
expected_loc = test_loc + '-entries-expected-win.json'
@@ -300,6 +289,7 @@ def test_extract_file_by_file_with_weird_names_7z(self):
def test_extract_file_by_file_weird_names_zip(self):
self.check_extract_file_by_file('sevenzip/weird_names.zip', regen=False)
+ @pytest.mark.xfail(on_windows, reason='Fails on Windows becasue it has file names that cannot be extracted there')
def test_extract_file_by_file_weird_names_ar(self):
self.check_extract_file_by_file('sevenzip/weird_names.ar', regen=False)