From c82f69195282bda1e5c2a6d756bebf42ce92ac9d Mon Sep 17 00:00:00 2001 From: funilrys Date: Wed, 28 Feb 2018 23:06:58 +0100 Subject: [PATCH 01/19] Review of get_file_by_url() Please note that this patch also introduce which is in charge of converting a domain in a line into IDNA and/or UTF-8 format. Also note the introduction of BeautifulSoup() which helps us to decode data from the downloaded URL. Fixes (issue(s)/protocol(s) I was able to reproduce): * https://github.com/StevenBlack/hosts/issues/514#issuecomment-368932152 Possible fix of (issue(s)/protocol(s) I wasn't able to reproduce): * https://github.com/StevenBlack/hosts/issues/514#issue-300048106 * https://github.com/StevenBlack/hosts/issues/494#issue-296166492 * https://github.com/StevenBlack/hosts/issues/420#issue-267453114 * https://github.com/StevenBlack/hosts/issues/372#issue-246927047 * https://github.com/StevenBlack/hosts/issues/382#issuecomment-322010562 --- updateHostsFile.py | 88 ++++++++++++++++++++++++++++++++++++++++------ 1 file changed, 78 insertions(+), 10 deletions(-) diff --git a/updateHostsFile.py b/updateHostsFile.py index ad77cc823..d27850ef8 100644 --- a/updateHostsFile.py +++ b/updateHostsFile.py @@ -6,23 +6,26 @@ # This Python script will combine all the host files you provide # as sources into one, unique host file to keep you internet browsing happy. -from __future__ import (absolute_import, division, - print_function, unicode_literals) -from glob import glob +from __future__ import (absolute_import, division, print_function, + unicode_literals) -import os +import argparse +import fnmatch +import json import locale +import os import platform import re import shutil +import socket import subprocess import sys import tempfile import time -import fnmatch -import argparse -import socket -import json +from glob import glob + +import lxml +from bs4 import BeautifulSoup # Detecting Python 3 for version-dependent implementations PY3 = sys.version_info >= (3, 0) @@ -1125,6 +1128,62 @@ def remove_old_hosts_file(backup): open(old_file_path, "a").close() # End File Logic +def domain_to_idna(line): + """ + Encode a domain which is presente into a line into `idna`. This way we avoid + the most encoding issue case. + + Parameters + ---------- + line : str + The line we have to encode/decode. + + Returns + ------- + line : str + The line in a converted format. + + Notes + ----- + - This method/function encode only the domain to `idna` format because in + most cases the encoding issue is due to a domain which looks like + `b'\xc9\xa2oogle.com'.decode('idna')`. + - About the splitting: + We split because we only want to encode the domain and not the full line + which may cause some issue. Keep in mind that we split but we still + concatenate once we encoded the domain. + + - The following split the prefix `0.0.0.0` or `127.0.0.1` of a line. + - The following also split the trailing comment of a given line. + - You do not get it ? + - Run https://git.io/vA1Rj and enjoy the view :-). + """ + + if not line.startswith('#'): + for separator in [' ', '\t']: + comment_to_append = '' + + if separator in line: + splited_line = line.split(separator) + if '#' in splited_line[1]: + comment_to_append = splited_line[1].split('#')[1] + + if comment_to_append: + splited_line[1] = splited_line[1] \ + .split(comment_to_append)[0] \ + .encode("IDNA").decode("UTF-8") + \ + '#' + comment_to_append[1] + else: + splited_line[1] = splited_line[1] \ + .encode("IDNA") \ + .decode("UTF-8") + '#' + else: + splited_line[1] = splited_line[1] \ + .encode("IDNA") \ + .decode("UTF-8") + return separator.join(splited_line) + return line.encode("IDNA").decode("UTF-8") + return line.encode("UTF-8").decode("UTF-8") # Helper Functions def get_file_by_url(url): @@ -1141,11 +1200,17 @@ def get_file_by_url(url): url_data : str or None The data retrieved at that URL from the file. Returns None if the attempted retrieval is unsuccessful. + + Note + ---- + - BeautifulSoup is used in this case to avoid having to search in which + format we have to encode or decode data before parsing it to UTF-8. """ try: f = urlopen(url) - return f.read().decode("UTF-8") + soup = BeautifulSoup(f.read(),'lxml').get_text() + return '\n'.join(list(map(domain_to_idna, soup.split('\n')))) except Exception: print("Problem getting file: ", url) @@ -1165,7 +1230,10 @@ def write_data(f, data): if PY3: f.write(bytes(data, "UTF-8")) else: - f.write(str(data).encode("UTF-8")) + try: + f.write(str(data)) + except UnicodeEncodeError: + f.write(str(data.encode("UTF-8"))) def list_dir_no_hidden(path): From ff58bbd1f2c903115e2d3a5df4fa83e20c23500f Mon Sep 17 00:00:00 2001 From: funilrys Date: Wed, 28 Feb 2018 23:08:45 +0100 Subject: [PATCH 02/19] Introduction of requirements.txt Please note that those file can be used to install dependencies with 'pip install -r requirements.txt' --- requirements.txt | 3 +++ requirements_python2.txt | 3 +++ 2 files changed, 6 insertions(+) create mode 100644 requirements.txt create mode 100644 requirements_python2.txt diff --git a/requirements.txt b/requirements.txt new file mode 100644 index 000000000..c51989f4f --- /dev/null +++ b/requirements.txt @@ -0,0 +1,3 @@ +lxml==4.1.1 +beautifulsoup4==4.6.0 +mock==2.0.0 diff --git a/requirements_python2.txt b/requirements_python2.txt new file mode 100644 index 000000000..5447edf82 --- /dev/null +++ b/requirements_python2.txt @@ -0,0 +1,3 @@ +mock==2.0.0 +lxml==4.1.1 +beautifulsoup4==4.6.0 From 079ad6b6743a34517ffc0214bb3f2bf2a69b20ff Mon Sep 17 00:00:00 2001 From: funilrys Date: Wed, 28 Feb 2018 23:13:13 +0100 Subject: [PATCH 03/19] Fix test issue. This patch fix https://travis-ci.org/funilrys/hosts/jobs/347497718#L748 --- updateHostsFile.py | 1 + 1 file changed, 1 insertion(+) diff --git a/updateHostsFile.py b/updateHostsFile.py index d27850ef8..7efb5c2c0 100644 --- a/updateHostsFile.py +++ b/updateHostsFile.py @@ -1128,6 +1128,7 @@ def remove_old_hosts_file(backup): open(old_file_path, "a").close() # End File Logic + def domain_to_idna(line): """ Encode a domain which is presente into a line into `idna`. This way we avoid From d3ef85df17425f1c6448cf2aa60cda267d24b2dc Mon Sep 17 00:00:00 2001 From: funilrys Date: Wed, 28 Feb 2018 23:15:01 +0100 Subject: [PATCH 04/19] Review typo + fix test issue. This patch fix https://travis-ci.org/funilrys/hosts/jobs/347497718#L749 --- updateHostsFile.py | 4 ++-- 1 file changed, 2 insertions(+), 2 deletions(-) diff --git a/updateHostsFile.py b/updateHostsFile.py index 7efb5c2c0..cf42d2542 100644 --- a/updateHostsFile.py +++ b/updateHostsFile.py @@ -1131,8 +1131,8 @@ def remove_old_hosts_file(backup): def domain_to_idna(line): """ - Encode a domain which is presente into a line into `idna`. This way we avoid - the most encoding issue case. + Encode a domain which is presente into a line into `idna`. This way we + avoid the most encoding issue. Parameters ---------- From 1fea720034db09597cc5eff82d3b6ff29472ba0b Mon Sep 17 00:00:00 2001 From: funilrys Date: Wed, 28 Feb 2018 23:20:01 +0100 Subject: [PATCH 05/19] Fix tests issue This patch fix https://travis-ci.org/funilrys/hosts/jobs/347500695#L398 --- updateHostsFile.py | 6 +++--- 1 file changed, 3 insertions(+), 3 deletions(-) diff --git a/updateHostsFile.py b/updateHostsFile.py index cf42d2542..c714d92d5 100644 --- a/updateHostsFile.py +++ b/updateHostsFile.py @@ -1150,9 +1150,9 @@ def domain_to_idna(line): most cases the encoding issue is due to a domain which looks like `b'\xc9\xa2oogle.com'.decode('idna')`. - About the splitting: - We split because we only want to encode the domain and not the full line - which may cause some issue. Keep in mind that we split but we still - concatenate once we encoded the domain. + We split because we only want to encode the domain and not the full + line which may cause some issue. Keep in mind that we split but we + still concatenate once we encoded the domain. - The following split the prefix `0.0.0.0` or `127.0.0.1` of a line. - The following also split the trailing comment of a given line. From 079d5ddd7f257c78ffd295743ff278626187be04 Mon Sep 17 00:00:00 2001 From: funilrys Date: Wed, 28 Feb 2018 23:22:32 +0100 Subject: [PATCH 06/19] Fix tests issue This patch fix https://travis-ci.org/funilrys/hosts/jobs/347500695#L397 Please also note that I introduced that patch because we do not directly use lxml but it is required by BeautifulSup() to parse the HTML. --- updateHostsFile.py | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/updateHostsFile.py b/updateHostsFile.py index c714d92d5..5e0002983 100644 --- a/updateHostsFile.py +++ b/updateHostsFile.py @@ -24,7 +24,7 @@ import tempfile import time from glob import glob -import lxml +import lxml # noqa: F401 from bs4 import BeautifulSoup # Detecting Python 3 for version-dependent implementations From f5c8ac58b281b224c9c7d71a121a2219c1b97872 Mon Sep 17 00:00:00 2001 From: funilrys Date: Wed, 28 Feb 2018 23:23:30 +0100 Subject: [PATCH 07/19] Fix tests issue. This patch fix https://travis-ci.org/funilrys/hosts/jobs/347500695#L399 --- updateHostsFile.py | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/updateHostsFile.py b/updateHostsFile.py index 5e0002983..952624735 100644 --- a/updateHostsFile.py +++ b/updateHostsFile.py @@ -1173,7 +1173,7 @@ def domain_to_idna(line): splited_line[1] = splited_line[1] \ .split(comment_to_append)[0] \ .encode("IDNA").decode("UTF-8") + \ - '#' + comment_to_append[1] + '#' + comment_to_append[1] else: splited_line[1] = splited_line[1] \ .encode("IDNA") \ From 3403b10e50a65fc7b076059894c1b91e62f073e0 Mon Sep 17 00:00:00 2001 From: funilrys Date: Wed, 28 Feb 2018 23:24:58 +0100 Subject: [PATCH 08/19] Fix tests issues. This patch fixes: * https://travis-ci.org/funilrys/hosts/jobs/347500695#L400 * https://travis-ci.org/funilrys/hosts/jobs/347500695#L401 --- updateHostsFile.py | 3 ++- 1 file changed, 2 insertions(+), 1 deletion(-) diff --git a/updateHostsFile.py b/updateHostsFile.py index 952624735..0482d8a60 100644 --- a/updateHostsFile.py +++ b/updateHostsFile.py @@ -1186,6 +1186,7 @@ def domain_to_idna(line): return line.encode("IDNA").decode("UTF-8") return line.encode("UTF-8").decode("UTF-8") + # Helper Functions def get_file_by_url(url): """ @@ -1210,7 +1211,7 @@ def get_file_by_url(url): try: f = urlopen(url) - soup = BeautifulSoup(f.read(),'lxml').get_text() + soup = BeautifulSoup(f.read(), 'lxml').get_text() return '\n'.join(list(map(domain_to_idna, soup.split('\n')))) except Exception: print("Problem getting file: ", url) From 1141823bc8ef36a18309c0011702d5a96e175335 Mon Sep 17 00:00:00 2001 From: funilrys Date: Wed, 28 Feb 2018 23:29:18 +0100 Subject: [PATCH 09/19] Fix tests issues. This patch introduce the installation of dependencies needed my the main commit. This patch fixes: * https://travis-ci.org/funilrys/hosts/jobs/347504195#L592 * https://travis-ci.org/funilrys/hosts/jobs/347504195#L598 --- ci/setup_conda_env.sh | 2 +- updateHostsFile.py | 2 +- 2 files changed, 2 insertions(+), 2 deletions(-) diff --git a/ci/setup_conda_env.sh b/ci/setup_conda_env.sh index cf3cfcb25..a5bb3e1bc 100755 --- a/ci/setup_conda_env.sh +++ b/ci/setup_conda_env.sh @@ -5,4 +5,4 @@ conda create -n hosts python=$PYTHON_VERSION || exit 1 source activate hosts echo "Installing packages..." -conda install mock flake8 +conda install mock flake8 beautifulsoup4 lxml diff --git a/updateHostsFile.py b/updateHostsFile.py index 0482d8a60..e28f0de5a 100644 --- a/updateHostsFile.py +++ b/updateHostsFile.py @@ -24,7 +24,7 @@ import tempfile import time from glob import glob -import lxml # noqa: F401 +import lxml # noqa: F401 from bs4 import BeautifulSoup # Detecting Python 3 for version-dependent implementations From 8f00cb4d762b1f99fbff0a5e5d18dac47c6f151e Mon Sep 17 00:00:00 2001 From: funilrys Date: Fri, 2 Mar 2018 21:43:52 +0100 Subject: [PATCH 10/19] Deletion of a trailing '#'. Please note that I have added that '#' by mistake. --- updateHostsFile.py | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/updateHostsFile.py b/updateHostsFile.py index e28f0de5a..0c78c3e2f 100644 --- a/updateHostsFile.py +++ b/updateHostsFile.py @@ -1177,7 +1177,7 @@ def domain_to_idna(line): else: splited_line[1] = splited_line[1] \ .encode("IDNA") \ - .decode("UTF-8") + '#' + .decode("UTF-8") else: splited_line[1] = splited_line[1] \ .encode("IDNA") \ From 780e47ffe5e7a2e978b7a6a42fd63b44a7e05ece Mon Sep 17 00:00:00 2001 From: funilrys Date: Fri, 2 Mar 2018 22:07:11 +0100 Subject: [PATCH 11/19] Review of domain_to_idna() This patch review the way we get the comment at the end of a line. I also did an application of DRY (Do not Repeat Yourself) and/or KISS (Keep It Simple, Stupid) by refactoring the 2 `else` statements into one line. --- updateHostsFile.py | 26 ++++++++++++-------------- 1 file changed, 12 insertions(+), 14 deletions(-) diff --git a/updateHostsFile.py b/updateHostsFile.py index 0c78c3e2f..29fa4e5b1 100644 --- a/updateHostsFile.py +++ b/updateHostsFile.py @@ -1161,27 +1161,25 @@ def domain_to_idna(line): """ if not line.startswith('#'): - for separator in [' ', '\t']: - comment_to_append = '' + for separator in ['\t', ' ']: + comment = '' if separator in line: splited_line = line.split(separator) if '#' in splited_line[1]: - comment_to_append = splited_line[1].split('#')[1] + index_comment = splited_line[1].find('#') + + if index_comment > -1: + comment = splited_line[1][index_comment:] - if comment_to_append: splited_line[1] = splited_line[1] \ - .split(comment_to_append)[0] \ + .split(comment)[0] \ .encode("IDNA").decode("UTF-8") + \ - '#' + comment_to_append[1] - else: - splited_line[1] = splited_line[1] \ - .encode("IDNA") \ - .decode("UTF-8") - else: - splited_line[1] = splited_line[1] \ - .encode("IDNA") \ - .decode("UTF-8") + comment + + splited_line[1] = splited_line[1] \ + .encode("IDNA") \ + .decode("UTF-8") return separator.join(splited_line) return line.encode("IDNA").decode("UTF-8") return line.encode("UTF-8").decode("UTF-8") From 4798710029b1cfc23130a0588c3a2c7e59171f86 Mon Sep 17 00:00:00 2001 From: funilrys Date: Fri, 2 Mar 2018 22:23:21 +0100 Subject: [PATCH 12/19] Introduction of `domain_to_idna()` tests. --- testUpdateHostsFile.py | 111 +++++++++++++++++++++++++++++++++++------ 1 file changed, 95 insertions(+), 16 deletions(-) diff --git a/testUpdateHostsFile.py b/testUpdateHostsFile.py index c1968c882..b4f4bfdca 100644 --- a/testUpdateHostsFile.py +++ b/testUpdateHostsFile.py @@ -5,25 +5,29 @@ # # Python script for testing updateHostFiles.py -from updateHostsFile import ( - Colors, PY3, colorize, display_exclusion_options, exclude_domain, - flush_dns_cache, gather_custom_exclusions, get_defaults, get_file_by_url, - is_valid_domain_format, matches_exclusions, move_hosts_file_into_place, - normalize_rule, path_join_robust, print_failure, print_success, - prompt_for_exclusions, prompt_for_move, prompt_for_flush_dns_cache, - prompt_for_update, query_yes_no, recursive_glob, remove_old_hosts_file, - supports_color, strip_rule, update_all_sources, update_readme_data, - update_sources_data, write_data, write_opening_header) - -import updateHostsFile -import unittest -import tempfile -import locale -import shutil import json -import sys +import locale import os import re +import shutil +import sys +import tempfile +import unittest + +import updateHostsFile +from updateHostsFile import (PY3, Colors, colorize, display_exclusion_options, + domain_to_idna, exclude_domain, flush_dns_cache, + gather_custom_exclusions, get_defaults, + get_file_by_url, is_valid_domain_format, + matches_exclusions, move_hosts_file_into_place, + normalize_rule, path_join_robust, print_failure, + print_success, prompt_for_exclusions, + prompt_for_flush_dns_cache, prompt_for_move, + prompt_for_update, query_yes_no, recursive_glob, + remove_old_hosts_file, strip_rule, supports_color, + update_all_sources, update_readme_data, + update_sources_data, write_data, + write_opening_header) if PY3: from io import BytesIO, StringIO @@ -1360,6 +1364,81 @@ def mock_url_open_decode_fail(_): return m +class DomainToIDNA(Base): + + def __init__(self, *args, **kwargs): + super(DomainToIDNA, self).__init__(*args, **kwargs) + + self.domains = [b'\xc9\xa2oogle.com', b'www.huala\xc3\xb1e.cl'] + self.expected_domains = ['xn--oogle-wmc.com', 'www.xn--hualae-0wa.cl'] + + def test_empty_line(self): + data = ["", "\r", "\n"] + + for empty in data: + expected = empty + + actual = domain_to_idna(empty) + self.assertEqual(actual, expected) + + def test_commented_line(self): + data = "# Hello World" + expected = data + actual = domain_to_idna(data) + + self.assertEqual(actual, expected) + + def test_simple_line(self): + # Test with a space as separator. + for i in range(len(self.domains)): + data = (b"0.0.0.0 " + self.domains[i]).decode('utf-8') + expected = "0.0.0.0 " + self.expected_domains[i] + + actual = domain_to_idna(data) + + self.assertEqual(actual, expected) + + # Test with a tabulation as separator. + for i in range(len(self.domains)): + data = (b"0.0.0.0\t" + self.domains[i]).decode('utf-8') + expected = "0.0.0.0\t" + self.expected_domains[i] + + actual = domain_to_idna(data) + + self.assertEqual(actual, expected) + + def test_single_line_with_comment_at_the_end(self): + # Test with a space as separator. + for i in range(len(self.domains)): + data = (b"0.0.0.0 " + self.domains[i] + b" # Hello World") \ + .decode('utf-8') + expected = "0.0.0.0 " + self.expected_domains[i] + " # Hello World" + + actual = domain_to_idna(data) + + self.assertEqual(actual, expected) + + # Test with a tabulation as separator. + for i in range(len(self.domains)): + data = (b"0.0.0.0\t" + self.domains[i] + b" # Hello World") \ + .decode('utf-8') + expected = "0.0.0.0\t" + self.expected_domains[i] + \ + " # Hello World" + + actual = domain_to_idna(data) + + self.assertEqual(actual, expected) + + def test_single_line_without_prefix(self): + for i in range(len(self.domains)): + data = self.domains[i].decode('utf-8') + expected = self.expected_domains[i] + + actual = domain_to_idna(data) + + self.assertEqual(actual, expected) + + class GetFileByUrl(BaseStdout): @mock.patch("updateHostsFile.urlopen", From d98b31fb921ff35fb1a421243fa4876504cf676e Mon Sep 17 00:00:00 2001 From: funilrys Date: Fri, 2 Mar 2018 22:43:24 +0100 Subject: [PATCH 13/19] Removing of `condescending` line. This patch fix : https://github.com/StevenBlack/hosts/pull/520/files/4798710029b1cfc23130a0588c3a2c7e59171f86#r171969863 --- updateHostsFile.py | 2 -- 1 file changed, 2 deletions(-) diff --git a/updateHostsFile.py b/updateHostsFile.py index 29fa4e5b1..7c1afab76 100644 --- a/updateHostsFile.py +++ b/updateHostsFile.py @@ -1156,8 +1156,6 @@ def domain_to_idna(line): - The following split the prefix `0.0.0.0` or `127.0.0.1` of a line. - The following also split the trailing comment of a given line. - - You do not get it ? - - Run https://git.io/vA1Rj and enjoy the view :-). """ if not line.startswith('#'): From d06bea8fb88bb259d05bacd33590b85dc973a746 Mon Sep 17 00:00:00 2001 From: funilrys Date: Fri, 2 Mar 2018 22:49:15 +0100 Subject: [PATCH 14/19] Introduction of dependencies installation instructions --- readme_template.md | 14 +++++++++----- 1 file changed, 9 insertions(+), 5 deletions(-) diff --git a/readme_template.md b/readme_template.md index 92148c5a6..57a7efab8 100644 --- a/readme_template.md +++ b/readme_template.md @@ -43,9 +43,13 @@ To run unit tests, in the top level directory, just run: python testUpdateHostsFile.py -**Note** if you are using Python 2, you must first install the `mock` library: +**Note** if you are using Python 2, please install the dependencies with: - pip install mock + pip install -r requirements_python2.txt + +**Note** if you are using Python 3, please install the dependencies with: + + pip install -r requirements.txt The `updateHostsFile.py` script, which is Python 2.7 and Python 3-compatible, will generate a unified hosts file based on the sources in the local `data/` @@ -104,9 +108,9 @@ in a subfolder. If the subfolder does not exist, it will be created. section at the top, containing lines like `127.0.0.1 localhost`. This is useful for configuring proximate DNS services on the local network. -`--compress`, or `-c`: `false` (default) or `true`, *Compress* the hosts file -ignoring non-necessary lines (empty lines and comments) and putting multiple -domains in each line. Reducing the number of lines of the hosts file improves +`--compress`, or `-c`: `false` (default) or `true`, *Compress* the hosts file +ignoring non-necessary lines (empty lines and comments) and putting multiple +domains in each line. Reducing the number of lines of the hosts file improves the performances under Windows (with DNS Client service enabled). `--minimise`, or `-m`: `false` (default) or `true`, like `--compress`, but puts From bebf7744caff333c487f2bdb491d3a4427319998 Mon Sep 17 00:00:00 2001 From: funilrys Date: Fri, 2 Mar 2018 22:51:10 +0100 Subject: [PATCH 15/19] Review of the `domain_to_idna()` notes. This patch fix : https://github.com/StevenBlack/hosts/pull/520#discussion_r171971574 --- updateHostsFile.py | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/updateHostsFile.py b/updateHostsFile.py index 7c1afab76..3231e954e 100644 --- a/updateHostsFile.py +++ b/updateHostsFile.py @@ -1146,7 +1146,7 @@ def domain_to_idna(line): Notes ----- - - This method/function encode only the domain to `idna` format because in + - This function encode only the domain to `idna` format because in most cases the encoding issue is due to a domain which looks like `b'\xc9\xa2oogle.com'.decode('idna')`. - About the splitting: From 6e62383b28a278e824eb70606dfc6d78d3bfdbfe Mon Sep 17 00:00:00 2001 From: funilrys Date: Fri, 2 Mar 2018 22:53:15 +0100 Subject: [PATCH 16/19] Review of Notes indentation This patch fix : https://github.com/StevenBlack/hosts/pull/520#discussion_r171971481 + It also fix (forgoten coma) : https://github.com/StevenBlack/hosts/pull/520#discussion_r171971574 --- updateHostsFile.py | 6 +++--- 1 file changed, 3 insertions(+), 3 deletions(-) diff --git a/updateHostsFile.py b/updateHostsFile.py index 3231e954e..a350ac08b 100644 --- a/updateHostsFile.py +++ b/updateHostsFile.py @@ -1147,12 +1147,12 @@ def domain_to_idna(line): Notes ----- - This function encode only the domain to `idna` format because in - most cases the encoding issue is due to a domain which looks like + most cases, the encoding issue is due to a domain which looks like `b'\xc9\xa2oogle.com'.decode('idna')`. - About the splitting: We split because we only want to encode the domain and not the full - line which may cause some issue. Keep in mind that we split but we - still concatenate once we encoded the domain. + line which may cause some issue. Keep in mind that we split but we + still concatenate once we encoded the domain. - The following split the prefix `0.0.0.0` or `127.0.0.1` of a line. - The following also split the trailing comment of a given line. From 50fde09ed7265863c23bef01d88ecccafbe37f84 Mon Sep 17 00:00:00 2001 From: funilrys Date: Fri, 2 Mar 2018 22:56:32 +0100 Subject: [PATCH 17/19] Fix grammar. This patch fix: https://github.com/StevenBlack/hosts/pull/520/files/d98b31fb921ff35fb1a421243fa4876504cf676e#r171971716 Thanks to @gfyoung --- updateHostsFile.py | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/updateHostsFile.py b/updateHostsFile.py index a350ac08b..e69cf3a8c 100644 --- a/updateHostsFile.py +++ b/updateHostsFile.py @@ -1151,7 +1151,7 @@ def domain_to_idna(line): `b'\xc9\xa2oogle.com'.decode('idna')`. - About the splitting: We split because we only want to encode the domain and not the full - line which may cause some issue. Keep in mind that we split but we + line, which may cause some issues. Keep in mind that we split, but we still concatenate once we encoded the domain. - The following split the prefix `0.0.0.0` or `127.0.0.1` of a line. From 93748698d5e7e923888e5079d2028d8d201ec9e3 Mon Sep 17 00:00:00 2001 From: funilrys Date: Sat, 3 Mar 2018 20:34:10 +0100 Subject: [PATCH 18/19] Update of readme_template.md Please note that this patch explitly set which `pip` version to use according to the user Python version. --- readme_template.md | 12 +++++++----- 1 file changed, 7 insertions(+), 5 deletions(-) diff --git a/readme_template.md b/readme_template.md index 57a7efab8..26952bfe4 100644 --- a/readme_template.md +++ b/readme_template.md @@ -43,13 +43,15 @@ To run unit tests, in the top level directory, just run: python testUpdateHostsFile.py -**Note** if you are using Python 2, please install the dependencies with: - - pip install -r requirements_python2.txt - **Note** if you are using Python 3, please install the dependencies with: - pip install -r requirements.txt + pip3 install --user -r requirements.txt + +**Note** if you are using Python 2, please install the dependencies with: + + pip2 install --user -r requirements_python2.txt + +**Note** we recommend the `--user` flag which installs the required dependencies at the user level. More information about it can be found on pip [documentation](https://pip.pypa.io/en/stable/reference/pip_install/?highlight=--user#cmdoption-user). The `updateHostsFile.py` script, which is Python 2.7 and Python 3-compatible, will generate a unified hosts file based on the sources in the local `data/` From 1e64d1287a65086365607fe540ad872b8fa92d64 Mon Sep 17 00:00:00 2001 From: funilrys Date: Sat, 3 Mar 2018 21:09:45 +0100 Subject: [PATCH 19/19] Review of readme_template.md Please note that this patch mothe the unit tests paragraph after dependencies installation. --- readme_template.md | 8 ++++---- 1 file changed, 4 insertions(+), 4 deletions(-) diff --git a/readme_template.md b/readme_template.md index 26952bfe4..6d9b8f737 100644 --- a/readme_template.md +++ b/readme_template.md @@ -39,10 +39,6 @@ folders. ## Generate your own unified hosts file -To run unit tests, in the top level directory, just run: - - python testUpdateHostsFile.py - **Note** if you are using Python 3, please install the dependencies with: pip3 install --user -r requirements.txt @@ -53,6 +49,10 @@ To run unit tests, in the top level directory, just run: **Note** we recommend the `--user` flag which installs the required dependencies at the user level. More information about it can be found on pip [documentation](https://pip.pypa.io/en/stable/reference/pip_install/?highlight=--user#cmdoption-user). +To run unit tests, in the top level directory, just run: + + python testUpdateHostsFile.py + The `updateHostsFile.py` script, which is Python 2.7 and Python 3-compatible, will generate a unified hosts file based on the sources in the local `data/` subfolder. The script will prompt you whether it should fetch updated