feat: implement knowledge management system (spec complete, all 53 tasks done)
This commit is contained in:
@@ -0,0 +1,187 @@
|
||||
"""Property-Based Tests für Quick Capture.
|
||||
|
||||
**Validates: Requirements 3.3, 8.3**
|
||||
|
||||
Property 7: Filename Pattern from Title
|
||||
*For any* non-empty title string, the Quick Capture filename generator SHALL
|
||||
produce a filename matching the pattern
|
||||
`\\d{4}-\\d{2}-\\d{2}-\\d{2}-\\d{2}-[a-z0-9-]+\\.md`
|
||||
(date-time prefix + slugified title + .md extension).
|
||||
|
||||
Property 11: URL Detection in Quick Capture
|
||||
*For any* string that is a valid HTTP or HTTPS URL, the Quick Capture input
|
||||
type detector SHALL classify it as type "url". For any string that is not a
|
||||
valid HTTP(S) URL, it SHALL NOT classify it as "url".
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import re
|
||||
from pathlib import Path
|
||||
|
||||
from hypothesis import given, settings, assume
|
||||
from hypothesis import strategies as st
|
||||
|
||||
from monorepo.knowledge.ingestion.capture import QuickCapture
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# Shared fixtures / helpers
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
FILENAME_PATTERN = re.compile(r"^\d{4}-\d{2}-\d{2}-\d{2}-\d{2}-[a-z0-9-]+\.md$")
|
||||
|
||||
|
||||
def make_capture() -> QuickCapture:
|
||||
"""Create a QuickCapture instance with a dummy root path."""
|
||||
return QuickCapture(Path("/tmp"))
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# Strategies
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
# Non-empty title strings with various characters (letters, digits, punctuation, spaces).
|
||||
# Filtered to contain at least one alphanumeric char, since the slugifier strips
|
||||
# all non-alnum chars — a title like "/" would yield an empty slug. In practice,
|
||||
# _derive_title always produces titles with alphanumeric content (from URLs,
|
||||
# file stems, or first 60 chars of text).
|
||||
title_st = st.text(
|
||||
alphabet=st.characters(
|
||||
whitelist_categories=("L", "N", "P", "Z"),
|
||||
blacklist_characters="\x00",
|
||||
),
|
||||
min_size=1,
|
||||
max_size=200,
|
||||
).filter(lambda s: s.strip() != "" and any(c.isalnum() for c in s))
|
||||
|
||||
# Valid HTTP/HTTPS URLs: protocol + domain + optional path
|
||||
_domain_st = st.from_regex(r"[a-z][a-z0-9]{1,20}\.[a-z]{2,6}", fullmatch=True)
|
||||
_path_segment_st = st.from_regex(r"[a-z0-9\-]{1,15}", fullmatch=True)
|
||||
_url_path_st = st.lists(_path_segment_st, min_size=0, max_size=3).map(
|
||||
lambda parts: "/" + "/".join(parts) if parts else ""
|
||||
)
|
||||
_protocol_st = st.sampled_from(["http://", "https://"])
|
||||
|
||||
valid_url_st = st.builds(
|
||||
lambda proto, domain, path: f"{proto}{domain}{path}",
|
||||
_protocol_st,
|
||||
_domain_st,
|
||||
_url_path_st,
|
||||
)
|
||||
|
||||
# Non-URL strings: plain text that does NOT start with http:// or https://
|
||||
non_url_st = st.text(
|
||||
alphabet=st.characters(
|
||||
whitelist_categories=("L", "N", "P", "Z"),
|
||||
blacklist_characters="\x00",
|
||||
),
|
||||
min_size=1,
|
||||
max_size=100,
|
||||
).filter(
|
||||
lambda s: not s.strip().startswith("http://")
|
||||
and not s.strip().startswith("https://")
|
||||
)
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# Property 7: Filename Pattern from Title
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
|
||||
class TestFilenamePatternFromTitle:
|
||||
"""**Validates: Requirements 3.3**
|
||||
|
||||
Property 7: For any non-empty title string, _generate_filename SHALL produce
|
||||
a filename matching the expected pattern with timestamp prefix, slugified
|
||||
title, and .md extension.
|
||||
"""
|
||||
|
||||
@given(title=title_st)
|
||||
@settings(max_examples=300)
|
||||
def test_filename_matches_pattern(self, title: str) -> None:
|
||||
"""Any non-empty title produces a filename matching the date-slug pattern."""
|
||||
capture = make_capture()
|
||||
filename = capture._generate_filename(title)
|
||||
assert FILENAME_PATTERN.match(filename), (
|
||||
f"Filename '{filename}' from title '{title!r}' "
|
||||
f"does not match pattern {FILENAME_PATTERN.pattern}"
|
||||
)
|
||||
|
||||
@given(title=title_st)
|
||||
@settings(max_examples=300)
|
||||
def test_slug_portion_max_50_chars(self, title: str) -> None:
|
||||
"""The slug portion of the filename never exceeds 50 characters."""
|
||||
capture = make_capture()
|
||||
filename = capture._generate_filename(title)
|
||||
# Filename format: YYYY-MM-DD-HH-MM-{slug}.md
|
||||
# Timestamp prefix is exactly "YYYY-MM-DD-HH-MM-" = 17 chars
|
||||
# Suffix is ".md" = 3 chars
|
||||
slug = filename[17:-3] # strip timestamp prefix and .md suffix
|
||||
assert len(slug) <= 50, (
|
||||
f"Slug '{slug}' has {len(slug)} chars, exceeds 50 limit"
|
||||
)
|
||||
|
||||
@given(title=title_st)
|
||||
@settings(max_examples=300)
|
||||
def test_filename_ends_with_md(self, title: str) -> None:
|
||||
"""Every generated filename ends with .md."""
|
||||
capture = make_capture()
|
||||
filename = capture._generate_filename(title)
|
||||
assert filename.endswith(".md")
|
||||
|
||||
@given(title=title_st)
|
||||
@settings(max_examples=200)
|
||||
def test_slug_has_no_trailing_hyphen(self, title: str) -> None:
|
||||
"""The slug portion does not end with a trailing hyphen."""
|
||||
capture = make_capture()
|
||||
filename = capture._generate_filename(title)
|
||||
# Remove .md suffix, then check the last char before .md
|
||||
without_md = filename[:-3]
|
||||
assert not without_md.endswith("-"), (
|
||||
f"Filename '{filename}' has trailing hyphen in slug"
|
||||
)
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# Property 11: URL Detection in Quick Capture
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
|
||||
class TestURLDetectionInQuickCapture:
|
||||
"""**Validates: Requirements 8.3**
|
||||
|
||||
Property 11: For any valid HTTP(S) URL, _detect_input_type SHALL return "url".
|
||||
For any non-URL string, it SHALL NOT return "url".
|
||||
"""
|
||||
|
||||
@given(url=valid_url_st)
|
||||
@settings(max_examples=300)
|
||||
def test_valid_urls_classified_as_url(self, url: str) -> None:
|
||||
"""Any string starting with http:// or https:// followed by a domain is 'url'."""
|
||||
capture = make_capture()
|
||||
result = capture._detect_input_type(url)
|
||||
assert result == "url", (
|
||||
f"URL '{url}' was classified as '{result}', expected 'url'"
|
||||
)
|
||||
|
||||
@given(text=non_url_st)
|
||||
@settings(max_examples=300)
|
||||
def test_non_urls_not_classified_as_url(self, text: str) -> None:
|
||||
"""Any string that is NOT a valid HTTP(S) URL SHALL NOT be classified as 'url'."""
|
||||
capture = make_capture()
|
||||
result = capture._detect_input_type(text)
|
||||
assert result != "url", (
|
||||
f"Non-URL text '{text!r}' was incorrectly classified as 'url'"
|
||||
)
|
||||
|
||||
@given(url=valid_url_st)
|
||||
@settings(max_examples=200)
|
||||
def test_urls_with_leading_whitespace_still_detected(self, url: str) -> None:
|
||||
"""URLs with leading/trailing whitespace are still classified as 'url'."""
|
||||
capture = make_capture()
|
||||
padded = f" {url} "
|
||||
result = capture._detect_input_type(padded)
|
||||
assert result == "url", (
|
||||
f"Padded URL '{padded}' was classified as '{result}', expected 'url'"
|
||||
)
|
||||
Reference in New Issue
Block a user