Initial commit. Basic featues are implemented
This commit is contained in:
+155
@@ -0,0 +1,155 @@
|
||||
"""Tracklist file parsing with flexible format support."""
|
||||
|
||||
import re
|
||||
|
||||
|
||||
def parse_format(format_str: str):
|
||||
"""
|
||||
Convert a format string (e.g., "%ts %tn - %an") into a list of tokens.
|
||||
|
||||
Each token is a dict with keys:
|
||||
'type': 'literal' or 'placeholder'
|
||||
'value': the literal text or the placeholder name (e.g., 'tn')
|
||||
|
||||
Placeholders are identified by a '%' followed by alphabetic characters.
|
||||
A literal '%%' is escaped to a single '%'.
|
||||
|
||||
Args:
|
||||
format_str: The user‑provided format string.
|
||||
|
||||
Returns:
|
||||
List of token dicts.
|
||||
|
||||
Raises:
|
||||
ValueError: If the format is invalid (e.g., unknown placeholder).
|
||||
"""
|
||||
# Known placeholders.
|
||||
valid_placeholders = {'ts', 'tn', 'an', 'al', 'date', 'ext'}
|
||||
|
||||
tokens = []
|
||||
i = 0
|
||||
while i < len(format_str):
|
||||
ch = format_str[i]
|
||||
if ch == '%':
|
||||
# Check for escaped '%%'.
|
||||
if i + 1 < len(format_str) and format_str[i + 1] == '%':
|
||||
tokens.append({'type': 'literal', 'value': '%'})
|
||||
i += 2
|
||||
continue
|
||||
# Must be a placeholder.
|
||||
# Match '%' followed by letters.
|
||||
match = re.match(r'%([a-zA-Z]+)', format_str[i:])
|
||||
if not match:
|
||||
raise ValueError(f"Invalid placeholder at position {i}: '{format_str[i:]}'")
|
||||
placeholder = match.group(1)
|
||||
if placeholder not in valid_placeholders:
|
||||
raise ValueError(f"Unknown placeholder '%{placeholder}'. "
|
||||
f"Allowed: {', '.join(valid_placeholders)}")
|
||||
tokens.append({'type': 'placeholder', 'value': placeholder})
|
||||
i += len(match.group(0))
|
||||
else:
|
||||
# Literal character.
|
||||
# Collect consecutive non‑'%' characters.
|
||||
j = i
|
||||
while j < len(format_str) and format_str[j] != '%':
|
||||
j += 1
|
||||
tokens.append({'type': 'literal', 'value': format_str[i:j]})
|
||||
i = j
|
||||
return tokens
|
||||
|
||||
|
||||
def parse_line(line: str, tokens):
|
||||
"""
|
||||
Parse a single line of the tracklist using the token list.
|
||||
|
||||
Returns a dict mapping placeholder names to extracted strings.
|
||||
If a placeholder is not found, its value is None.
|
||||
|
||||
Args:
|
||||
line: The raw line from the tracklist file.
|
||||
tokens: The token list from parse_format().
|
||||
|
||||
Returns:
|
||||
A dict with keys for each placeholder present in the format.
|
||||
|
||||
Raises:
|
||||
ValueError: If the line cannot be parsed according to the format.
|
||||
"""
|
||||
# Strip leading/trailing whitespace but preserve internal spaces.
|
||||
line = line.strip()
|
||||
if not line:
|
||||
raise ValueError("Empty line")
|
||||
|
||||
result = {}
|
||||
pos = 0
|
||||
# We need to match the tokens in order.
|
||||
for token in tokens:
|
||||
if token['type'] == 'literal':
|
||||
literal = token['value']
|
||||
if not line.startswith(literal, pos):
|
||||
raise ValueError(f"Expected literal '{literal}' at position {pos}, got '{line[pos:]}'")
|
||||
pos += len(literal)
|
||||
else: # placeholder
|
||||
placeholder = token['value']
|
||||
# If this is the last token, capture the rest of the line.
|
||||
if token is tokens[-1]:
|
||||
result[placeholder] = line[pos:].strip() or None
|
||||
pos = len(line)
|
||||
else:
|
||||
# Find the next literal to use as a delimiter.
|
||||
# We need to look ahead to the next literal token.
|
||||
next_literal = None
|
||||
for next_token in tokens[tokens.index(token) + 1:]:
|
||||
if next_token['type'] == 'literal':
|
||||
next_literal = next_token['value']
|
||||
break
|
||||
if next_literal is None:
|
||||
# If no more literals, capture the rest.
|
||||
result[placeholder] = line[pos:].strip() or None
|
||||
pos = len(line)
|
||||
else:
|
||||
# Find the occurrence of the next literal in the line starting from pos.
|
||||
next_pos = line.find(next_literal, pos)
|
||||
if next_pos == -1:
|
||||
raise ValueError(f"Could not find literal '{next_literal}' after placeholder '{placeholder}'")
|
||||
result[placeholder] = line[pos:next_pos].strip() or None
|
||||
pos = next_pos
|
||||
return result
|
||||
|
||||
|
||||
def read_tracklist(filepath: str, tokens):
|
||||
"""
|
||||
Read and parse the tracklist file using the provided token list.
|
||||
|
||||
Each line is parsed into a dict of fields. The timestamp field ('ts')
|
||||
is required; if missing, an error is raised.
|
||||
|
||||
Args:
|
||||
filepath: Path to the tracklist file.
|
||||
tokens: Token list from parse_format().
|
||||
|
||||
Returns:
|
||||
A list of dicts, one per track.
|
||||
|
||||
Raises:
|
||||
ValueError: If a line cannot be parsed or the timestamp is missing.
|
||||
"""
|
||||
tracks = []
|
||||
with open(filepath, 'r', encoding='utf-8') as file:
|
||||
for line_num, raw_line in enumerate(file, 1):
|
||||
line = raw_line.strip()
|
||||
if not line:
|
||||
continue
|
||||
try:
|
||||
parsed = parse_line(line, tokens)
|
||||
except ValueError as error:
|
||||
print(f"Warning: Skipping line {line_num}: {error}")
|
||||
continue
|
||||
|
||||
# Ensure timestamp is present.
|
||||
if 'ts' not in parsed or parsed['ts'] is None:
|
||||
raise ValueError(f"Line {line_num}: timestamp (%ts) is missing or empty")
|
||||
|
||||
tracks.append(parsed)
|
||||
|
||||
return tracks
|
||||
Reference in New Issue
Block a user