Implementing link checking

This commit is contained in:
bhuvy
2019-07-23 14:12:57 -05:00
parent eb5019ccd1
commit 207754d371
7 changed files with 81 additions and 20 deletions
+4 -1
View File
@@ -1,6 +1,9 @@
dist: xenial
language: python
cache: pip
cache:
pip: true
directories:
- $HOME/cache/
branches:
only:
- master
+6
View File
@@ -43,6 +43,11 @@ github_shim = 'github_redefinitions.tex'
sed_regex = r'0,/\\\[1\\\]\\\[\\\]/{//d;}'
# Do all ops in the /tmp directory
tmp_dir = '/tmp/'
# Cache file directory
cache_file_dir = os.path.join(os.path.expanduser('~'), 'cache')
os.makedirs(cache_file_dir, exist_ok=True)
cache_file_name = os.path.join(cache_file_dir, 'link_cache.json')
class ConvertableTexFile(object):
"""
@@ -113,6 +118,7 @@ def generate_tex_meta(order, outdir, meta_file_name, chapter=None):
# Where to output the data
os.environ['META_FILE_NAME'] = meta_file_name
os.environ['LINK_CACHE_FILE_NAME'] = cache_file_name
# Order file does not have suffixes
logger.info("Generating Metadata at {}".format(meta_file_name))
+64 -1
View File
@@ -4,15 +4,21 @@
Pandoc filter to grab the different header levels as yaml to stderr
"""
from panflute import run_filter, Str, Header, MetaMap
from panflute import run_filter, Str, Header, MetaMap, Image, Link
import sys
import atexit
import yaml
import os
import os.path
try:
from yaml import CLoader as Loader, CDumper as Dumper
except ImportError:
from yaml import Loader, Dumper
import dateutil.parser
import datetime
import requests
link_cache_days = 30
# Metadata gleaned from the file
meta = dict(
@@ -24,6 +30,22 @@ meta = dict(
# Max header level
max_level = 2
# Cache of all the links
cache_file = os.environ['LINK_CACHE_FILE_NAME']
if not os.path.isfile(cache_file):
link_cache = dict()
else:
with open(cache_file, 'r') as f:
link_cache = yaml.load(f, Loader=Loader)
default_image_alt = 'image'
class NoAltTagException(Exception):
pass
class BadLinkException(Exception):
pass
def exit_handler():
"""
Loads a metadata file in the environment variable
@@ -43,6 +65,9 @@ def exit_handler():
with open(meta_file_name, 'w') as f:
f.write(out)
with open(cache_file, 'w+') as f:
f.write(yaml.dump(link_cache, Dumper=Dumper))
def deserialize(x):
"""
Takes a panflute element x and returns
@@ -53,13 +78,32 @@ def deserialize(x):
return x.text
return ' '
def is_valid_cache_info(link_info):
if link_info is None:
return False
date = dateutil.parser.parse(link_info)
time_between = datetime.datetime.now() - date
return time_between.days <= link_cache_days
def output_yaml(elem, doc):
"""
This function performs a walk along a document and outputs to
stderr a yaml file that has metadata about the name and the
subsections of the chapter. Elem is a pandoc valid element
doc is usually going to be None
This also validates all images and links within a cache
"""
if type(elem) == Image:
# Get the number of chars for the alt tag
alt_name = ''.join(map(deserialize, elem._content.list))
alt_length = len(elem._content)
# No alt means no compile
# Accessibility by default
if alt_length == 0 or alt_name.lower() == default_image_alt:
raise NoAltTagException(elem.url)
if type(elem) == Header and elem.level <= max_level:
# Format what the title is going to look like
name = ''.join(map(deserialize, elem.content.list))
@@ -74,6 +118,25 @@ def output_yaml(elem, doc):
bib_file = deserialize(dictionary_map['bibliography'][0].content[0])
meta['bib_file'] = bib_file
if isinstance(elem, Link):
url = elem.url
link_info = link_cache.get(url)
# Ignore internal links for the time being
# There are a lot of edge cases
if not url.startswith('#') and not is_valid_cache_info(link_info):
if not (url.startswith('http://') or url.startswith('https://')):
raise BadLinkException(url)
print('Requesting url "{}"'.format(url), file=sys.stderr)
# Ping the image
head = requests.head(url)
if head.status_code < 200 and head.status_code > 299:
raise BadLinkException(url)
# Otherwise reset the cache
link_cache[url] = datetime.datetime.now().isoformat()
return elem
def main(doc=None):
+2 -14
View File
@@ -11,10 +11,6 @@ import os.path
base_raw_url = 'https://raw.githubusercontent.com/illinois-cs241/coursebook/master/'
eps_ext = '.eps'
default_image_alt = 'image'
class NoAltTagException(Exception):
pass
def replace_suffix(content, suffix_old, suffix_new):
ret = content
@@ -34,17 +30,9 @@ def deserialize(x):
def doc_filter(elem, doc):
if type(elem) == Image:
# Get the number of chars for the alt tag
alt_name = ''.join(map(deserialize, elem._content.list))
alt_length = len(elem._content)
# No alt means no compile
# Accessibility by default
if alt_length == 0 or alt_name.lower() == default_image_alt:
raise NoAltTagException(elem.url)
# Otherwise link to the raw user link instead of relative
# Link to the raw user link instead of relative
# That way the wiki and the site will have valid links automagically
new_url = replace_suffix(elem.url, '.eps', '.png')
new_url = replace_suffix(elem.url, eps_ext, '.png')
if not os.path.isfile(new_url):
raise ValueError('{} Not found'.format(new_url))
elem.url = base_raw_url + new_url
+1 -1
View File
@@ -214,7 +214,7 @@ return;
\item \keyword{if else else-if} are control flow keywords.
There are a few ways to use these (1) A bare if (2) An if with an else (3) an if with an else-if (4) an if with an else if and else.
Note that an else is matched with the most recent if.
A subtle bug related to a mismatched if and else statement, is the \href{dangling else problem}{https://en.wikipedia.org/wiki/Dangling\_else}.
A subtle bug related to a mismatched if and else statement, is the \href{https://en.wikipedia.org/wiki/Dangling\_else}{dangling else problem}.
The statements are always executed from the if to the else.
If any of the intermediate conditions are true, the if block performs that action and goes to the end of that block.
+1
View File
@@ -1,3 +1,4 @@
panflute==1.10.6
PyYAML>=4.2b1
Jinja2>=2.10.1
requests==2.18.4
+3 -3
View File
@@ -1291,7 +1291,7 @@ Thread \#1 arrives, \emph{sets the turn back to 1} and now waits until thread 2
Yes
With a bit of searching, it is possible to find it in production for specific simple mobile processors today.
Peterson's algorithm is used to implement low-level Linux Kernel locks for the Tegra mobile processor (a system-on-chip ARM process and GPU core by Nvidia) \href{Link to Lock Source}{https://android.googlesource.com/kernel/tegra.git/+/android-tegra-3.10/arch/arm/mach-tegra/sleep.S\#58}
Peterson's algorithm is used to implement low-level Linux Kernel locks for the Tegra mobile processor (a system-on-chip ARM process and GPU core by Nvidia) \href{https://android.googlesource.com/kernel/tegra.git/+/android-tegra-3.10/arch/arm/mach-tegra/sleep.S\#58}{Link to Lock Source}
In general now, CPUs and C compilers can re-order CPU instructions or use CPU-core-specific local cache values that are stale if another core updates the shared variables.
Thus a simple pseudo-code to C implementation is too naive for most platforms.
@@ -1323,8 +1323,8 @@ Thus, in practice, surrounding critical sections with a mutex lock and unlock ca
For further reading, we suggest the following web post that discusses implementing Peterson's algorithm on an x86 process and the Linux documentation on memory barriers.
\begin{enumerate}
\item \href{Memory Fences}{http://bartoszmilewski.com/2008/11/05/who-ordered-memory-fences-on-an-x86/}
\item \href{Memory Barriers}{http://lxr.free-electrons.com/source/Documentation/memory-barriers.txt}
\item \href{http://bartoszmilewski.com/2008/11/05/who-ordered-memory-fences-on-an-x86/}{Memory Fences}
\item \href{http://lxr.free-electrons.com/source/Documentation/memory-barriers.txt}{Memory Barriers}
\end{enumerate}
\section{Implementing Counting Semaphore}