Created
February 9, 2019 17:57
-
-
Save Yavor-Ivanov/e3acdbf30c09448b7bf63423c725f3ba to your computer and use it in GitHub Desktop.
A script that searches a Cinera index file
This file contains hidden or bidirectional Unicode text that may be interpreted or compiled differently than what appears below. To review, open the file in an editor that reveals hidden Unicode characters.
Learn more about bidirectional Unicode characters
| #!/bin/env python | |
| from __future__ import print_function | |
| from pprint import pprint | |
| from math import floor | |
| from sys import argv, exit | |
| from enum import Enum as enum | |
| from termcolor import cprint, colored as coloured | |
| import re, argparse, textwrap | |
| from itertools import izip_longest as zip_longest | |
| # @Todo <Yavor>: | |
| # - @Check if the text wrapping is as good as we can get it. I think textwrap wraps the first line too short. | |
| # - @Check maybe just grab the terminal columns and default to that for wrapping? | |
| # - Add a way to read the parameters from the terminal env and from a config file. | |
| # | |
| # - Add a parse mode that tries to validate the index file. | |
| # - Have different keywords show up in different colours? | |
| # - Have the searh update its code.index file when you make a new search. | |
| # - Add some tests, maybe. | |
| class attr_dict(dict): | |
| def __init__(self, *args, **kwargs): | |
| super(attr_dict, self).__init__(*args, **kwargs) | |
| self.__dict__ = self | |
| def annotation_match(annotation, matched_keywords): | |
| return attr_dict(hour=annotation.hour, minute=annotation.minute, second=annotation.second, text=annotation.text, matched_keywords=matched_keywords) | |
| def video_match(video, matched_keywords, annotation_matches=[]): | |
| return attr_dict(name=video.name, title=video.title, annotations=annotation_matches, matched_keywords=matched_keywords) | |
| def video(name='', title=''): | |
| return attr_dict(name=name, title=title, annotations=[]) | |
| def annotation(hour, minute, second, text): | |
| return attr_dict(hour=hour, minute=minute, second=second, text=text) | |
| def unquote(str): | |
| return str.strip('"').strip('\'') | |
| def read_annotations(file): | |
| videos = [] | |
| mode_type = enum('mode_type', 'video_header annotation') | |
| with open(file) as f: | |
| mode = None | |
| for line in f: | |
| line = line.strip() | |
| if line == '---': | |
| mode = mode_type.video_header | |
| videos.append(video()) | |
| continue | |
| elif line.startswith('markers'): | |
| mode = mode_type.annotation | |
| continue | |
| if mode == mode_type.video_header: | |
| if line.startswith('name'): | |
| videos[-1].name = unquote(line.split(' ', 1)[1]) | |
| if line.startswith('title'): | |
| videos[-1].title = unquote(line.split(' ', 1)[1]) | |
| if mode == mode_type.annotation: | |
| raw_time = int(unquote(line.split(' ', 1)[0][:-1])) | |
| hour = floor(raw_time/60/60) | |
| minute = floor(raw_time/60) % 60 | |
| second = raw_time % 60 | |
| text = unquote(line.split(' ', 1)[1]) | |
| videos[-1].annotations.append(annotation(hour, minute, second, text)) | |
| return videos | |
| def search_match(needles, haystack): | |
| if not type(needles) in (tuple, list): | |
| needles = [needles] | |
| return [needle for needle in needles if needle.lower() in haystack.lower()] | |
| def search_annotations(videos, terms=[], only_title=False): | |
| result = [] | |
| for video in videos: | |
| title_match = search_match(terms, video.title) | |
| # @Speed <Yavor>: Stop executing search_match twice! | |
| annotations = [annotation_match(a, []) for a in video.annotations] if only_title else [annotation_match(a, search_match(terms, a.text)) for a in video.annotations if search_match(terms, a.text)] | |
| if (not only_title and annotations) or title_match: | |
| result.append(video_match(video, title_match, annotations)) | |
| return result | |
| def colourise_matches(input, matches, fg_colour=None, bg_colour=None, attrs=[]): | |
| result = input | |
| for match in matches: | |
| match_case = re.search(match, input, re.IGNORECASE).group() | |
| result = result.replace(match_case, coloured(match_case, fg_colour, bg_colour, attrs)) | |
| return result | |
| def print_search_results(videos, decoration, output_format='hierarchical', wrap_column=0): | |
| def output_annotation(text, wrap_column, decoration): | |
| if wrap_column > 0: | |
| text_parts = textwrap.wrap(text, width=wrap_column, subsequent_indent=(' '*15), initial_indent=(' '*4)) | |
| else: | |
| text_parts = [' ' + text] | |
| # @Cleanup <Yavor>: Fix this so it doesn't output the lines, but just wraps them! Also, rename the function to something meaningful. | |
| [print(part) for part in text_parts] | |
| # print('[%02d:%02d:%02d] %s' % (a.hour, a.minute, a.second, colourise_matches(a.text, a.matched_keywords, **decoration))) | |
| for video in videos: | |
| if output_format == 'hierarchical': | |
| print('Name: %s' % video.name) | |
| print('Title: %s' % colourise_matches(video.title, video.matched_keywords, **decoration)) | |
| # [print(' [%02d:%02d:%02d] %s' % (a.hour, a.minute, a.second, colourise_matches(a.text, a.matched_keywords, **decoration))) for a in video.annotations] | |
| [output_annotation('[%02d:%02d:%02d] %s' % (a.hour, a.minute, a.second, colourise_matches(a.text, a.matched_keywords, **decoration)), wrap_column, decoration) for a in video.annotations] | |
| print() | |
| elif output_format == 'title-list': | |
| print('%s: %s' % (video.name, colourise_matches(video.title, video.matched_keywords, **decoration))) | |
| if __name__ == '__main__': | |
| search_result_decoration = dict(attrs=['reverse']) | |
| parser = argparse.ArgumentParser(description='Search for keywords in a Cinera annotations file.') | |
| parser.add_argument('file', metavar='f', help='Cinera index file to search.') | |
| parser.add_argument('search', nargs='+', help="Text to search for.") | |
| parser.add_argument('--nocolor', action='store_true', default=False, help="Don't colour the output. (Useful if you want to do your own processing of the output)") | |
| parser.add_argument('--only-title', action='store_true', default=False, help="Don't match against the annotations, only match by the video title.") | |
| parser.add_argument('--wrap', type=int, default=0, help="Wrap text to N columns. When set to 0 (the default value), doesn't do wrapping.") | |
| parser.add_argument( | |
| '--output', default='hierarchical', choices=['hierarchical', 'title-list'], | |
| help="Output formatting to apply. 'Hierarchical' prints all annotations under the video name and title, 'title-list' prints a flat list of video titles" | |
| ) | |
| args = parser.parse_args() | |
| input_file = args.file | |
| search_terms = args.search | |
| if args.nocolor: | |
| search_result_decoration = dict() | |
| videos = read_annotations(input_file) | |
| search_results = search_annotations(videos, search_terms, args.only_title) | |
| print_search_results(search_results, search_result_decoration, args.output, args.wrap) |
Sign up for free
to join this conversation on GitHub.
Already have an account?
Sign in to comment