|
- # -*- coding: utf-8 -*-
-
- import re
- import sys
- from string import punctuation
- from typing import List, Tuple, Dict
- from urllib.parse import urlparse
-
- # Temporary replacement
- # The descriptions that contain () at the end must adapt to the new policy later
- punctuation = punctuation.replace('()', '')
-
- anchor = '###'
- auth_keys = ['apiKey', 'OAuth', 'X-Mashape-Key', 'User-Agent', 'No']
- https_keys = ['Yes', 'No']
- cors_keys = ['Yes', 'No', 'Unknown']
-
- index_title = 0
- index_desc = 1
- index_auth = 2
- index_https = 3
- index_cors = 4
- index_call = 5
-
- num_segments = 6
- min_segments = 5
- max_segments = 6
- min_entries_per_category = 3
- max_description_length = 100
-
- anchor_re = re.compile(anchor + r'\s(.+)')
- category_title_in_index_re = re.compile(r'\*\s\[(.*)\]')
- link_re = re.compile(r'\[(.+)\]\((http.*)\)')
-
- # Type aliases
- APIList = List[str]
- Categories = Dict[str, APIList]
- CategoriesLineNumber = Dict[str, int]
-
-
- def error_message(line_number: int, message: str) -> str:
- line = line_number + 1
- return f'(L{line:03d}) {message}'
-
-
- def get_categories_content(contents: List[str]) -> Tuple[Categories, CategoriesLineNumber]:
-
- categories = {}
- category_line_num = {}
-
- for line_num, line_content in enumerate(contents):
-
- if line_content.startswith(anchor):
- category = line_content.split(anchor)[1].strip()
- categories[category] = []
- category_line_num[category] = line_num
- continue
-
- if not line_content.startswith('|') or line_content.startswith('|---'):
- continue
-
- raw_title = [
- raw_content.strip() for raw_content in line_content.split('|')[1:-1]
- ][0]
-
- title_match = link_re.match(raw_title)
- if title_match:
- title = title_match.group(1).upper()
- categories[category].append(title)
-
- return (categories, category_line_num)
-
-
- def check_alphabetical_order(lines: List[str]) -> List[str]:
-
- err_msgs = []
-
- categories, category_line_num = get_categories_content(contents=lines)
-
- for category, api_list in categories.items():
- if sorted(api_list) != api_list:
- err_msg = error_message(
- category_line_num[category],
- f'{category} category is not alphabetical order'
- )
- err_msgs.append(err_msg)
-
- return err_msgs
-
-
- def check_title(line_num: int, raw_title: str) -> List[str]:
-
- err_msgs = []
-
- title_match = link_re.match(raw_title)
-
- # url should be wrapped in "[TITLE](LINK)" Markdown syntax
- if not title_match:
- err_msg = error_message(line_num, 'Title syntax should be "[TITLE](LINK)"')
- err_msgs.append(err_msg)
- else:
- # do not allow "... API" in the entry title
- title = title_match.group(1)
- if title.upper().endswith(' API'):
- err_msg = error_message(line_num, 'Title should not end with "... API". Every entry is an API here!')
- err_msgs.append(err_msg)
-
- return err_msgs
-
-
- def check_description(line_num: int, description: str) -> List[str]:
-
- err_msgs = []
-
- first_char = description[0]
- if first_char.upper() != first_char:
- err_msg = error_message(line_num, 'first character of description is not capitalized')
- err_msgs.append(err_msg)
-
- last_char = description[-1]
- if last_char in punctuation:
- err_msg = error_message(line_num, f'description should not end with {last_char}')
- err_msgs.append(err_msg)
-
- desc_length = len(description)
- if desc_length > max_description_length:
- err_msg = error_message(line_num, f'description should not exceed {max_description_length} characters (currently {desc_length})')
- err_msgs.append(err_msg)
-
- return err_msgs
-
-
- def check_auth(line_num: int, auth: str) -> List[str]:
-
- err_msgs = []
-
- backtick = '`'
- if auth != 'No' and (not auth.startswith(backtick) or not auth.endswith(backtick)):
- err_msg = error_message(line_num, 'auth value is not enclosed with `backticks`')
- err_msgs.append(err_msg)
-
- if auth.replace(backtick, '') not in auth_keys:
- err_msg = error_message(line_num, f'{auth} is not a valid Auth option')
- err_msgs.append(err_msg)
-
- return err_msgs
-
-
- def check_https(line_num: int, https: str) -> List[str]:
-
- err_msgs = []
-
- if https not in https_keys:
- err_msg = error_message(line_num, f'{https} is not a valid HTTPS option')
- err_msgs.append(err_msg)
-
- return err_msgs
-
-
- def check_cors(line_num: int, cors: str) -> List[str]:
-
- err_msgs = []
-
- if cors not in cors_keys:
- err_msg = error_message(line_num, f'{cors} is not a valid CORS option')
- err_msgs.append(err_msg)
-
- return err_msgs
-
- def extract_url(markdown_link: str) -> str:
- match = re.search(r'\((http[^)]+)\)', markdown_link)
- return match.group(1) if match else ''
-
- def uri_validator(url):
- try:
- result = urlparse(url)
- return all([result.scheme, result.netloc]) or ' '
- except ValueError:
- return False
-
- def check_calls(line_num: int, calls: str) -> List[str]:
-
- err_msgs = []
-
- if not uri_validator(calls):
- err_msg = error_message(line_num, 'Call This API column must contain a valid URL')
- err_msgs.append(err_msg)
- else:
- actual_url = extract_url(calls)
- parsed_url = urlparse(actual_url)
- if not parsed_url.netloc.endswith('pstmn.io') and not parsed_url.netloc.endswith('postman.com'):
- err_msg = error_message(line_num, 'Call This API column URL must be a run in Postman button')
- err_msgs.append(err_msg)
- return err_msgs
-
- def check_entry(line_num: int, segments: List[str]) -> List[str]:
-
- raw_title = segments[index_title]
- description = segments[index_desc]
- auth = segments[index_auth]
- https = segments[index_https]
- cors = segments[index_cors]
-
- title_err_msgs = check_title(line_num, raw_title)
- desc_err_msgs = check_description(line_num, description)
- auth_err_msgs = check_auth(line_num, auth)
- https_err_msgs = check_https(line_num, https)
- cors_err_msgs = check_cors(line_num, cors)
-
- err_msgs = [
- *title_err_msgs,
- *desc_err_msgs,
- *auth_err_msgs,
- *https_err_msgs,
- *cors_err_msgs,
- ]
-
- if len(segments) == max_segments:
- calls_column = segments[index_call].strip()
- if calls_column:
- optional_column_err_msgs = check_calls(line_num, calls_column)
- err_msgs.extend(optional_column_err_msgs)
-
-
- return err_msgs
-
-
- def check_file_format(lines: List[str]) -> List[str]:
-
- err_msgs = []
- category_title_in_index = []
-
- alphabetical_err_msgs = check_alphabetical_order(lines)
- err_msgs.extend(alphabetical_err_msgs)
-
- num_in_category = min_entries_per_category + 1
- category = ''
- category_line = 0
-
- # Flag to indicate whether we are in the main content section
- in_main_content = False
-
- for line_num, line_content in enumerate(lines):
- # Check if the line marks the start of the main content section
- if "## Index" in line_content:
- in_main_content = True
- continue
-
- # Skip lines until we reach the main content section
- if not in_main_content:
- continue
-
- category_title_match = category_title_in_index_re.match(line_content)
- if category_title_match:
- category_title_in_index.append(category_title_match.group(1))
-
- # check each category for the minimum number of entries
- if line_content.startswith(anchor):
- category_match = anchor_re.match(line_content)
- if category_match:
- if category_match.group(1) not in category_title_in_index:
- err_msg = error_message(line_num, f'category header ({category_match.group(1)}) not added to Index section')
- err_msgs.append(err_msg)
- else:
- err_msg = error_message(line_num, 'category header is not formatted correctly')
- err_msgs.append(err_msg)
-
- if num_in_category < min_entries_per_category:
- err_msg = error_message(category_line, f'{category} category does not have the minimum {min_entries_per_category} entries (only has {num_in_category})')
- err_msgs.append(err_msg)
-
- category = line_content.split(' ')[1]
- category_line = line_num
- num_in_category = 0
- continue
-
- # skips lines that we do not care about
- if not line_content.startswith('|') or line_content.startswith('|:---'):
- continue
-
- num_in_category += 1
- segments = line_content.split('|')[1:-1]
- if len(segments) < 5 or len(segments) > 6:
- err_msg = error_message(line_num, f'entry does not have all the required columns (have {len(segments)}, need {min_segments} to {max_segments})')
- err_msgs.append(err_msg)
- continue
-
- for segment in segments:
- # every line segment should start and end with exactly 1 space
- if len(segment) - len(segment.lstrip()) != 1 or len(segment) - len(segment.rstrip()) != 1:
- err_msg = error_message(line_num, 'each segment must start and end with exactly 1 space')
- err_msgs.append(err_msg)
-
- segments = [segment.strip() for segment in segments]
- entry_err_msgs = check_entry(line_num, segments)
- err_msgs.extend(entry_err_msgs)
-
- return err_msgs
-
-
- def main(filename: str) -> None:
-
- with open(filename, mode='r', encoding='utf-8') as file:
- lines = list(line.rstrip() for line in file)
-
- file_format_err_msgs = check_file_format(lines)
-
- if file_format_err_msgs:
- for err_msg in file_format_err_msgs:
- print(err_msg)
- sys.exit(1)
-
-
- if __name__ == '__main__':
-
- num_args = len(sys.argv)
-
- if num_args < 2:
- print('No .md file passed (file should contain Markdown table syntax)', flush=True)
- sys.exit(1)
-
- filename = sys.argv[1]
-
- main(filename)
|