Skip to content

Instantly share code, notes, and snippets.

@Yavor-Ivanov
Created February 9, 2019 17:57
Show Gist options
  • Select an option

  • Save Yavor-Ivanov/e3acdbf30c09448b7bf63423c725f3ba to your computer and use it in GitHub Desktop.

Select an option

Save Yavor-Ivanov/e3acdbf30c09448b7bf63423c725f3ba to your computer and use it in GitHub Desktop.
A script that searches a Cinera index file
#!/bin/env python
from __future__ import print_function
from pprint import pprint
from math import floor
from sys import argv, exit
from enum import Enum as enum
from termcolor import cprint, colored as coloured
import re, argparse, textwrap
from itertools import izip_longest as zip_longest
# @Todo <Yavor>:
# - @Check if the text wrapping is as good as we can get it. I think textwrap wraps the first line too short.
# - @Check maybe just grab the terminal columns and default to that for wrapping?
# - Add a way to read the parameters from the terminal env and from a config file.
#
# - Add a parse mode that tries to validate the index file.
# - Have different keywords show up in different colours?
# - Have the searh update its code.index file when you make a new search.
# - Add some tests, maybe.
class attr_dict(dict):
def __init__(self, *args, **kwargs):
super(attr_dict, self).__init__(*args, **kwargs)
self.__dict__ = self
def annotation_match(annotation, matched_keywords):
return attr_dict(hour=annotation.hour, minute=annotation.minute, second=annotation.second, text=annotation.text, matched_keywords=matched_keywords)
def video_match(video, matched_keywords, annotation_matches=[]):
return attr_dict(name=video.name, title=video.title, annotations=annotation_matches, matched_keywords=matched_keywords)
def video(name='', title=''):
return attr_dict(name=name, title=title, annotations=[])
def annotation(hour, minute, second, text):
return attr_dict(hour=hour, minute=minute, second=second, text=text)
def unquote(str):
return str.strip('"').strip('\'')
def read_annotations(file):
videos = []
mode_type = enum('mode_type', 'video_header annotation')
with open(file) as f:
mode = None
for line in f:
line = line.strip()
if line == '---':
mode = mode_type.video_header
videos.append(video())
continue
elif line.startswith('markers'):
mode = mode_type.annotation
continue
if mode == mode_type.video_header:
if line.startswith('name'):
videos[-1].name = unquote(line.split(' ', 1)[1])
if line.startswith('title'):
videos[-1].title = unquote(line.split(' ', 1)[1])
if mode == mode_type.annotation:
raw_time = int(unquote(line.split(' ', 1)[0][:-1]))
hour = floor(raw_time/60/60)
minute = floor(raw_time/60) % 60
second = raw_time % 60
text = unquote(line.split(' ', 1)[1])
videos[-1].annotations.append(annotation(hour, minute, second, text))
return videos
def search_match(needles, haystack):
if not type(needles) in (tuple, list):
needles = [needles]
return [needle for needle in needles if needle.lower() in haystack.lower()]
def search_annotations(videos, terms=[], only_title=False):
result = []
for video in videos:
title_match = search_match(terms, video.title)
# @Speed <Yavor>: Stop executing search_match twice!
annotations = [annotation_match(a, []) for a in video.annotations] if only_title else [annotation_match(a, search_match(terms, a.text)) for a in video.annotations if search_match(terms, a.text)]
if (not only_title and annotations) or title_match:
result.append(video_match(video, title_match, annotations))
return result
def colourise_matches(input, matches, fg_colour=None, bg_colour=None, attrs=[]):
result = input
for match in matches:
match_case = re.search(match, input, re.IGNORECASE).group()
result = result.replace(match_case, coloured(match_case, fg_colour, bg_colour, attrs))
return result
def print_search_results(videos, decoration, output_format='hierarchical', wrap_column=0):
def output_annotation(text, wrap_column, decoration):
if wrap_column > 0:
text_parts = textwrap.wrap(text, width=wrap_column, subsequent_indent=(' '*15), initial_indent=(' '*4))
else:
text_parts = [' ' + text]
# @Cleanup <Yavor>: Fix this so it doesn't output the lines, but just wraps them! Also, rename the function to something meaningful.
[print(part) for part in text_parts]
# print('[%02d:%02d:%02d] %s' % (a.hour, a.minute, a.second, colourise_matches(a.text, a.matched_keywords, **decoration)))
for video in videos:
if output_format == 'hierarchical':
print('Name: %s' % video.name)
print('Title: %s' % colourise_matches(video.title, video.matched_keywords, **decoration))
# [print(' [%02d:%02d:%02d] %s' % (a.hour, a.minute, a.second, colourise_matches(a.text, a.matched_keywords, **decoration))) for a in video.annotations]
[output_annotation('[%02d:%02d:%02d] %s' % (a.hour, a.minute, a.second, colourise_matches(a.text, a.matched_keywords, **decoration)), wrap_column, decoration) for a in video.annotations]
print()
elif output_format == 'title-list':
print('%s: %s' % (video.name, colourise_matches(video.title, video.matched_keywords, **decoration)))
if __name__ == '__main__':
search_result_decoration = dict(attrs=['reverse'])
parser = argparse.ArgumentParser(description='Search for keywords in a Cinera annotations file.')
parser.add_argument('file', metavar='f', help='Cinera index file to search.')
parser.add_argument('search', nargs='+', help="Text to search for.")
parser.add_argument('--nocolor', action='store_true', default=False, help="Don't colour the output. (Useful if you want to do your own processing of the output)")
parser.add_argument('--only-title', action='store_true', default=False, help="Don't match against the annotations, only match by the video title.")
parser.add_argument('--wrap', type=int, default=0, help="Wrap text to N columns. When set to 0 (the default value), doesn't do wrapping.")
parser.add_argument(
'--output', default='hierarchical', choices=['hierarchical', 'title-list'],
help="Output formatting to apply. 'Hierarchical' prints all annotations under the video name and title, 'title-list' prints a flat list of video titles"
)
args = parser.parse_args()
input_file = args.file
search_terms = args.search
if args.nocolor:
search_result_decoration = dict()
videos = read_annotations(input_file)
search_results = search_annotations(videos, search_terms, args.only_title)
print_search_results(search_results, search_result_decoration, args.output, args.wrap)
Sign up for free to join this conversation on GitHub. Already have an account? Sign in to comment