Initial commit. Basic featues are implemented

This commit is contained in:
2026-07-28 16:25:11 +05:00
commit 842e3bf1b0
12 changed files with 1190 additions and 0 deletions
+155
View File
@@ -0,0 +1,155 @@
"""Tracklist file parsing with flexible format support."""
import re
def parse_format(format_str: str):
"""
Convert a format string (e.g., "%ts %tn - %an") into a list of tokens.
Each token is a dict with keys:
'type': 'literal' or 'placeholder'
'value': the literal text or the placeholder name (e.g., 'tn')
Placeholders are identified by a '%' followed by alphabetic characters.
A literal '%%' is escaped to a single '%'.
Args:
format_str: The userprovided format string.
Returns:
List of token dicts.
Raises:
ValueError: If the format is invalid (e.g., unknown placeholder).
"""
# Known placeholders.
valid_placeholders = {'ts', 'tn', 'an', 'al', 'date', 'ext'}
tokens = []
i = 0
while i < len(format_str):
ch = format_str[i]
if ch == '%':
# Check for escaped '%%'.
if i + 1 < len(format_str) and format_str[i + 1] == '%':
tokens.append({'type': 'literal', 'value': '%'})
i += 2
continue
# Must be a placeholder.
# Match '%' followed by letters.
match = re.match(r'%([a-zA-Z]+)', format_str[i:])
if not match:
raise ValueError(f"Invalid placeholder at position {i}: '{format_str[i:]}'")
placeholder = match.group(1)
if placeholder not in valid_placeholders:
raise ValueError(f"Unknown placeholder '%{placeholder}'. "
f"Allowed: {', '.join(valid_placeholders)}")
tokens.append({'type': 'placeholder', 'value': placeholder})
i += len(match.group(0))
else:
# Literal character.
# Collect consecutive non'%' characters.
j = i
while j < len(format_str) and format_str[j] != '%':
j += 1
tokens.append({'type': 'literal', 'value': format_str[i:j]})
i = j
return tokens
def parse_line(line: str, tokens):
"""
Parse a single line of the tracklist using the token list.
Returns a dict mapping placeholder names to extracted strings.
If a placeholder is not found, its value is None.
Args:
line: The raw line from the tracklist file.
tokens: The token list from parse_format().
Returns:
A dict with keys for each placeholder present in the format.
Raises:
ValueError: If the line cannot be parsed according to the format.
"""
# Strip leading/trailing whitespace but preserve internal spaces.
line = line.strip()
if not line:
raise ValueError("Empty line")
result = {}
pos = 0
# We need to match the tokens in order.
for token in tokens:
if token['type'] == 'literal':
literal = token['value']
if not line.startswith(literal, pos):
raise ValueError(f"Expected literal '{literal}' at position {pos}, got '{line[pos:]}'")
pos += len(literal)
else: # placeholder
placeholder = token['value']
# If this is the last token, capture the rest of the line.
if token is tokens[-1]:
result[placeholder] = line[pos:].strip() or None
pos = len(line)
else:
# Find the next literal to use as a delimiter.
# We need to look ahead to the next literal token.
next_literal = None
for next_token in tokens[tokens.index(token) + 1:]:
if next_token['type'] == 'literal':
next_literal = next_token['value']
break
if next_literal is None:
# If no more literals, capture the rest.
result[placeholder] = line[pos:].strip() or None
pos = len(line)
else:
# Find the occurrence of the next literal in the line starting from pos.
next_pos = line.find(next_literal, pos)
if next_pos == -1:
raise ValueError(f"Could not find literal '{next_literal}' after placeholder '{placeholder}'")
result[placeholder] = line[pos:next_pos].strip() or None
pos = next_pos
return result
def read_tracklist(filepath: str, tokens):
"""
Read and parse the tracklist file using the provided token list.
Each line is parsed into a dict of fields. The timestamp field ('ts')
is required; if missing, an error is raised.
Args:
filepath: Path to the tracklist file.
tokens: Token list from parse_format().
Returns:
A list of dicts, one per track.
Raises:
ValueError: If a line cannot be parsed or the timestamp is missing.
"""
tracks = []
with open(filepath, 'r', encoding='utf-8') as file:
for line_num, raw_line in enumerate(file, 1):
line = raw_line.strip()
if not line:
continue
try:
parsed = parse_line(line, tokens)
except ValueError as error:
print(f"Warning: Skipping line {line_num}: {error}")
continue
# Ensure timestamp is present.
if 'ts' not in parsed or parsed['ts'] is None:
raise ValueError(f"Line {line_num}: timestamp (%ts) is missing or empty")
tracks.append(parsed)
return tracks