From fd33a2144ac78223bbe8281776b043d61121b74d Mon Sep 17 00:00:00 2001 From: Jarvis Jr Hermes Date: Fri, 14 Aug 2026 12:10:47 +0000 Subject: [PATCH] feat: publish vssa-clients skill --- .gitea/workflows/validate.yml | 14 + README.md | 24 +- SKILL.md | 307 ++++++++++++++++++ references/cross-client-system-registry.md | 45 +++ references/cvp-is-organization-route.md | 62 ++++ references/derived-contractor-registry.md | 46 +++ references/lithuanian-source-routes.md | 57 ++++ ...ystem-infrastructure-lifecycle-analysis.md | 135 ++++++++ scripts/validate_skill.py | 22 ++ 9 files changed, 711 insertions(+), 1 deletion(-) create mode 100644 .gitea/workflows/validate.yml create mode 100644 SKILL.md create mode 100644 references/cross-client-system-registry.md create mode 100644 references/cvp-is-organization-route.md create mode 100644 references/derived-contractor-registry.md create mode 100644 references/lithuanian-source-routes.md create mode 100644 references/system-infrastructure-lifecycle-analysis.md create mode 100644 scripts/validate_skill.py diff --git a/.gitea/workflows/validate.yml b/.gitea/workflows/validate.yml new file mode 100644 index 0000000..590a7c4 --- /dev/null +++ b/.gitea/workflows/validate.yml @@ -0,0 +1,14 @@ +name: Validate skill + +on: + push: + branches: [main] + pull_request: + +jobs: + validate: + runs-on: ubuntu-latest + steps: + - uses: actions/checkout@v4 + - name: Validate skill package + run: python3 scripts/validate_skill.py diff --git a/README.md b/README.md index bbb6ca3..cd2faef 100644 --- a/README.md +++ b/README.md @@ -1,3 +1,25 @@ # vssa-clients -Hermes skill for evidence-backed VSSA client, system, cloud, contractor, and procurement research. \ No newline at end of file +Version-controlled Hermes skill for VSSA client, system, cloud, contractor, and procurement research. + +## Source of truth + +- Gitea: +- Branch: `main` +- Runtime name: `vssa-clients` +- Runtime path: `/opt/data/skills/research/vssa-clients` +- Maintainer worktree: `/opt/data/vssa-clients-skill` + +The runtime path is a symbolic link to this checked-out repository. This prevents a local skill copy from drifting away from Gitea. + +## Update workflow + +1. Fetch and fast-forward `main` before editing. +2. Modify `SKILL.md` or linked files in this repository. +3. Run `python3 scripts/validate_skill.py`. +4. Review `git diff --check` and the complete diff. +5. Commit and push to `main`. +6. Verify local/remote SHA equality and authenticated Gitea readback of `SKILL.md` plus a changed linked file. +7. Verify `skill_view(name="vssa-clients")` and the VSSA cron attachment. + +Do not patch a detached runtime-only copy. Client-specific research belongs in the VSSA documentation repository; only reusable investigation procedures belong here. diff --git a/SKILL.md b/SKILL.md new file mode 100644 index 0000000..e29e882 --- /dev/null +++ b/SKILL.md @@ -0,0 +1,307 @@ +--- +name: vssa-clients +description: "Use when investigating an organization’s information systems, cloud services, digital platforms, and implementation or maintenance contractors from public evidence. Produces one concise evidence-backed dossier per client and improves the investigation method after each recurring batch." +version: 1.0.0 +author: Hermes Agent +license: MIT +platforms: [linux, macos, windows] +metadata: + hermes: + tags: [osint, clients, systems, cloud, contractors, procurement, evidence] + related_skills: [gitea-repository-operations] +--- + +# VSSA Clients + +## Overview + +## Authoritative repository + +This skill is managed in Gitea at `https://gitea.lego-cloud.eu/vssa-v1-skills-code-agent/vssa-clients`. Treat the repository `main` branch as the source of truth. For every reusable improvement, edit the repository worktree at `/opt/data/vssa-clients-skill`, validate the complete package, commit and push it, verify remote readback, and only then report the skill update. The runtime installation is a symlink to that worktree; do not maintain a divergent local-only copy. + +Investigate one organization at a time and produce a concise file answering four questions: + +1. What does the organization do? +2. Which systems, portals, registries, SaaS products, infrastructure, or clouds does it operate, procure, use, or depend on? +3. What does each identified system do? +4. Which company or public-sector contractor built, implemented, hosts, supports, or modernizes it? + +Every material claim must link directly to evidence. Unknown values stay `Not publicly identified`; never turn a technical hint into a factual attribution. + +## Required Output + +Create one Markdown file per client. Prefer a compact table over repeated narrative. + +```markdown +# — + +- **Purpose:** +- **Status:** researched | partial | no-public-evidence +- **Last checked:** YYYY-MM-DD + +## Systems, clouds and contractors + +| System / service | Meaning | Client relationship | Contractor / company | Evidence | Confidence | +|---|---|---|---|---|---| +| | | operates / owns / procures / uses / depends on | | [source](direct URL) | high / medium / low | + +## Gaps + +- +``` + +Keep the file concise: + +- one row per materially distinct system or service; +- one-sentence purpose; +- no duplicated `Findings`, `Evidence`, and `Search record` sections; +- attach the strongest direct citation to each row; +- add a second citation only when it independently proves a different part of the row; +- do not pad a dossier with generic websites, social media, or unsupported technology guesses. + +## Research Workflow + +### 1. Establish identity + +Record the exact legal name, former names, English rendering, abbreviations, parent ministry, institution code when available, and official domains. Former names matter because contracts and vendor case studies may use historical identities. + +Before attributing any system, perform an **identity-collision check**: compare the exact legal name, institution code, parent body, and official domain in the source with the target client. Treat punctuation, diacritics, and acronym variants as potentially material. Similar abbreviations can denote unrelated institutions (especially where one acronym differs only by a Lithuanian diacritic), so a system catalogue on a similarly named organization's domain must be excluded unless another authoritative source proves the relationship. + +### 2. Establish purpose + +Use the institution’s regulations, official mandate page, or founding legal act. Compress the mandate to one sentence; do not reproduce statutory text. + +### 3. Find named systems + +Search the exact organization name and its variants with Lithuanian and English system terms: + +- `"" "informacinė sistema"` +- `"" sistema OR registras OR portalas OR platforma` +- `site: (sistema OR registras OR portalas OR e. paslaugos)` +- `site: filetype:pdf (sistemos OR architektūra OR saugos nuostatai)` +- `"" (cloud OR debesis OR debesija OR SaaS OR IaaS OR PaaS)` +- `"" (modernizavimas OR diegimas OR kūrimas OR palaikymas)` + +Check menus, service catalogues, privacy notices, cookie notices, accessibility statements, system regulations, security policies, API/integration documentation, project pages, annual reports, and open-data dataset metadata. On Lithuanian state-archive `*.archyvai.lrv.lt` homepages, inspect the visible `Informacinės sistemos` / `Paslaugos` block and its raw links: it commonly exposes EAIS staff, institution and digital-reading-room endpoints plus archive-specific public collections even when ordinary navigation search is sparse. The archive page proves that institution's use or operation; a central-platform vendor project page proves only the historical platform role, not an institution-specific contract or current support/hosting. On Lithuanian `*.lrv.lt` sites, inspect `/sitemap.xml` directly: project-page slugs often expose exact system names and lead to official pages with modernization scope, dates, funding, and client roles even when site navigation or external search is weak. When an institution's current domain is unknown or an old hostname no longer resolves, inspect its supervising ministry's sitemap for `valdymo-srities-istaigos` entries: the official institution entry may redirect to the current subdomain and simultaneously corroborate the exact institutional identity. Open and cite the resolved institution page; the sitemap remains discovery evidence only. On WordPress sites, `/sitemap.xml` or `/wp-sitemap.xml` may be only a sitemap index; open each referenced child sitemap—especially `wp-sitemap-posts-page-*.xml`—and inspect both page and attachment URLs for terms such as `dienynas`, `sistema`, `prisijungimas`, `mokiniams`, `paslaugos`, and policy-document filenames. Direct PDF attachments can expose director-approved electronic-diary rules that ordinary navigation omits. Extract the full PDF text and inspect embedded link annotations where available: a dated official procedure may explicitly name the institution, SaaS product, provider, operational purpose, and login URL. Treat it as evidence at the document date, not proof of an unchanged current subscription, maintenance arrangement, or production host. Then open the resulting official page or document and any relevant provider page; the sitemap itself is discovery evidence, not claim evidence. If an ordinary text/link extractor finds the visible product label but not its destination, inspect the raw HTML around that label: school and institution sites often wrap a product-logo image in an otherwise textless anchor, hiding the decisive external SaaS URL from text-only extraction. Also inspect audience-specific utility pages such as `Mokytojams`, `Darbuotojams`, `Mokiniams`, and `Prisijungimai`: a single official page may expose direct links to an electronic diary, Microsoft 365, webmail, and a shared public-sector document system even when the homepage and sitemap contain none of those product names. Read each anchor destination rather than relying only on its generic visible label. The institution page proves the institution's use; pair it with the product's legal/provider page or an official public-sector system catalogue to establish the provider or service-operator role. Keep that role bounded: a catalogue entry showing that an agency provides or manages a shared system does not by itself prove that agency developed it, hosts the institution's tenant, supplies current local support, or holds an institution-specific contract. On Lithuanian government sites, inspect `/lt/paslaugos/` as a compact system catalogue: it may explicitly name several portals, describe what each does, and link to their operational endpoints. Group services that share one named backend instead of turning every individual e-service form into a separate system row. If a previously cited official deep service URL now returns `404`, do not assume the service disappeared or preserve the dead citation: search the site's current `/sitemap.xml` for distinctive words from the old title, system acronym, or service label, then inspect both Lithuanian and English catalogue variants. Open the discovered replacement page, confirm its body still supports the claim, and cite the replacement rather than the sitemap. This is especially useful when a site redesign moves an old nested e-service page into a top-level service catalogue or language-specific system page. On Lithuanian judiciary sites, a central migration or outage notice republished on an individual court's own official domain can directly prove that court's dependency on a shared system when the body gives that court's users account, case, document, payment, or continuity instructions. Use the court-hosted copy for the court relationship and the central administration page only for central strategy or support roles. Conversely, site-wide footer/sidebar links to systems such as a judicial portal or personnel system are catalogue leads only: on a subunit page they do not prove that the subunit uses, owns, or procures every linked service. When external search is weak, also try the official institution or municipality site's own search endpoint (commonly `/search?q=`), then open the resulting article rather than citing the search page. If the institution's site or operational endpoint remains bot-blocked after one browser-equivalent retry, search the exact institution and system name on its supervising ministry's official domain: ministry policy FAQs and service explanations can provide an opened, authoritative description of the institution-system relationship. Cite that ministry page rather than the inaccessible endpoint, and do not extend its claim to suppliers, hosting, or current support unless it says so explicitly. Privacy notices often name processors and SaaS products; accessibility declarations often reveal the formal system owner and platform name. + +### 4. Investigate contractors deeply + +For every named system, run a contractor query ladder using the full name, acronym, and client name: + +- `"" (rangovas OR tiekėjas OR vykdytojas OR diegėjas OR kūrėjas)` +- `"" (sutartis OR pirkimas OR laimėtojas OR pasiūlymas)` +- `"" (maintenance OR support OR implementation OR modernization)` +- `"" "" (UAB OR AB OR MB OR consortium)` +- `site:*.lt "" (klientas OR projektas OR case study)` +- `site:ted.europa.eu ""` +- exact phrases from tender titles, technical specifications, acceptance announcements, and vendor portfolios. + +Search both directions: + +1. **Client → system → procurement/contractor** through official and procurement records. +2. **System → vendor → client** through contractor portfolios, case studies, press releases, staff biographies, conference talks, and project lists. + +Contractors commonly place client names in `Projects`, `Clients`, `Case studies`, `Portfolio`, `News`, or downloadable PDF capability statements. Search the contractor’s site for both the institution and system acronym. + +For school SaaS and electronic diaries, inspect the institution homepage/header/footer for its external login link, then open the linked service's login page and legal footer. The institution link can prove product use while the service footer can explicitly identify the software company. Together they support a provider attribution, but they do not prove contract dates, procurement route, separate implementation/support suppliers, or production hosting. + +Also inspect the official institution site's own footer for credits such as `Sprendimas: UAB ...` (solution by). This can directly identify the public website or service-portal solution provider when the credit is visible in the opened page body. Limit the attribution to that portal solution: the credit does not establish contract dates, current maintenance, hosting, or any protected back-office system. + +Apply the same two-link test to academic-library discovery services: an institution's official library link can prove use of its named virtual catalogue, while the platform vendor's official product page can identify the software licensor/provider. Record that attribution as medium confidence unless a contract or implementation record independently confirms the institution-specific arrangement; neither the vendor-owned hostname nor the product page alone proves a local integrator, contract dates, support supplier, or the hosting of unrelated institutional systems. + +Apply a two-link test to hosted ticketing platforms as well: the institution's official ticket instructions establish dependency on the named distributor, while the distributor's current legal terms or privacy page can identify the platform operator and any group company explicitly assigned an IT/data-processing role. Keep the attribution at medium confidence unless an institution-specific contract or award is found; vendor terms do not establish local contract dates, implementation scope, support allocation, or production hosting for the institution's other systems. Ticketing sites may redirect an obsolete legal-page slug to a current company-specific path, so cite the opened final URL after confirming the body. + +Apply the same bounded two-link method to hosted media-distribution platforms such as YouTube or Spotify: the institution's official recordings/media page must link its named channel or artist profile to prove use, while the platform's current legal terms can identify the platform operator. Attribute only the platform-provider role. Do not infer the institution's distributor, rights-management supplier, paid subscription, contract dates, support allocation, or the hosting of original institutional media from those two links. Treat a generic social-media icon or site-wide footer link as a discovery lead unless the official page clearly identifies the channel as institutional and ties it to the relevant media service. + +Identify the role precisely: developer, implementation partner, hosting provider, cloud provider, maintenance supplier, software licensor, integrator, subcontractor, or consortium member. Do not collapse these into a generic “contractor.” Include contract or project dates when available so historical work is not presented as current. + +For legacy public e-service projects, pair the institution's dated project page with its current electronic-service catalogue when both exist. The project page may explicitly identify the original creation supplier and planned completion date, while the current catalogue independently proves that the institution still provides the service. Attribute the company only as the historical developer/creation supplier; current availability does not establish current maintenance, support, hosting, or an ongoing contract. + +Treat a named permanent on-site interactive or digital exhibition as a distinct system/service when an official institution or funding-body source describes its digital components—such as video, audio, screens, sensors, projections, or interactive installations—and explicitly ties it to the client. Keep the installation as one observation rather than splitting every device or thematic station into separate systems. A dated official account saying that the new exhibition had opened and recorded visitors during a named month directly establishes production operation no later than that month; record only month precision and do not substitute the article's later publication date as the release date. This does not establish original delivery kickoff, hosting classification, or a later software release. If the source names only an artistic or curator-led team as creator, do not turn that informal team into a contractor-registry entity or infer a contracting company, integrator, or support supplier; keep the contractor `Not publicly identified` and preserve the creator detail as bounded context. + +### 5. Use procurement and funding evidence + +For repeatable Lithuanian discovery routes—including official-site internal search, the `data.gov.lt` organization/dataset fallback for bot-blocked institutions, personal-data controller/processor wording, direct VPT document handling, and pre-delivery link checks—see [`references/lithuanian-source-routes.md`](references/lithuanian-source-routes.md). + +Search: + +- CPVA's official search portal at `https://cpva.lt/paieska` using exact client legal names, former names, system names/acronyms, and project/digitalization terms. Inspect the live form rather than assuming a conventional query parameter: the confirmed Search & Filter result URL uses `?_sf_s=` (see [`references/lithuanian-source-routes.md`](references/lithuanian-source-routes.md)). A `200` response to an invented `?q=`, `?search=`, or ineffective POST may be only the unfiltered search page. CPVA uses broad token matching, so even a populated result list can be irrelevant. Treat results and snippets as discovery only; open the CPVA article or linked document and require its body to prove the exact client/system relationship, project, date, funding, supplier, or delivery role before citing it; +- Lithuania’s public-procurement sources and contracting-authority publications; +- TED, the EU Tenders Electronic Daily portal: https://ted.europa.eu/en/; +- the institution’s procurement plans, technical specifications, award notices, contract registers, and annual reports; +- e-Seimas and e-TAR for system regulations, owner/processor assignments, and legal-name history; +- ES investment and Recovery and Resilience project pages for implementers, partners, budgets, deliverables, and dates; +- data.gov.lt metadata for system owners, dataset providers, and update responsibilities; +- EU and national funding portals where project consortia or suppliers are explicitly named. + +A tender notice proves intended procurement, not necessarily award or delivery. Prefer an award notice, signed contract, completion announcement, acceptance record, or corroborated vendor case study for contractor attribution. + +### 5a. Learn and use Lithuania’s CVP IS portal safely + +For the VSSA continuous-research workflow, `https://viesiejipirkimai.lt/epps/viewCFTSAction.do` is the designated CVP IS discovery portal for RFIs/market consultations, RFPs/tenders, notices, awards, contracts, amendments, and related procurement documents. Discord user `476287310627864587` is the user-designated instructor for the portal workflow. Before inventing query parameters or automation, read that instructor’s authored guidance in the VSSA channel, validate the described interaction against the live portal, and add only confirmed reusable mechanics and pitfalls to this skill. Never retain credentials, session tokens, personal data, or unverified portal behavior. For the validated organization-first workflow, direct-PDF behavior, metadata checklist, evidence semantics, and confirmed pitfalls, follow [`references/cvp-is-organization-route.md`](references/cvp-is-organization-route.md). + +A second validated discovery route is **organization first**: + +1. Open `https://viesiejipirkimai.lt/epps/viewOrganisations.do`, select/use the **Organizacija** search, and search the exact client legal name plus known former-name/abbreviation variants. If the byte-exact registry name returns no result, retry a small, documented set of typography/legal-form variants—for example, remove Lithuanian quotation marks or expand/shorten `Viešoji įstaiga` / `VšĮ` / `AB`—because CVP IS may store a different legal-form rendering. Parse complete result rows and bind each displayed organization name to the profile link in that same row; do not associate links by a broad surrounding-HTML substring. Do not use a broader substring match as identity evidence. +2. Identity-check the result row before opening it. Preserve the query variant that matched, portal organization ID, exact displayed organization name, and available registration number/domain; reconcile these against the authoritative client identity before proceeding. If a nominally exact query returns multiple distinct rows or IDs, treat it as unresolved until registration number, domain, or another authoritative identifier disambiguates the target—never choose the first row. This remains true when the profiles show the same address or appear to be duplicate historical/current records. Open and review every exact candidate profile and its notice list, preserve each `(organizationId, authorityId, orgGroupId)` tuple separately, and report the duplicate-profile ambiguity rather than silently designating one canonical record. Do not map a near-name match merely because it has purchases. +3. Open the organization profile at `https://viesiejipirkimai.lt/epps/prepareViewCAOrganisation.do?id=`. +4. Follow **PERŽIŪRĖTI VISUS PASKELBTUS SKELBIMUS**. The resulting URL has the validated shape `https://viesiejipirkimai.lt/epps/notices/viewPublishedNotices.do?authorityId=&orgGroupId=`. Obtain both IDs from the profile link rather than assuming they are equal or deriving one arithmetically. +5. Review all result pages, increasing the page size when possible. Parse complete table rows rather than only anchors: CVP IS commonly links the notice type in the first cell while rendering the procurement title as plain text in the second cell, so anchor-only extraction can miss relevant IT titles. Derive any `d--p=` pagination key from the live page instead of hard-coding the table identifier. Capture notice type, procurement title, upload/publication dates, language, and status. Search the titles for exact system names/acronyms and IT/digitalization terms, but treat every row as discovery only. +6. The notice-type link may return a direct `application/pdf` attachment with `Content-Disposition: attachment`, which browser navigation can report as aborted because it downloads rather than renders. Fetch the exact link as a file, verify the MIME type and size, preserve the full parameterized URL and notice identifiers, extract the PDF text/OCR where necessary, and classify the document by its body before making any claim. + +The organization notice list is an additional route, not a complete procurement history: it lists published notices associated with that portal organization and may omit other stages, historical migrations, contracts, amendments, or records published under predecessor/parent organizations. Continue the direct procurement/system searches and former-name checks as well. + +When mapping a procurement record, preserve the exact contracting authority and identity-check it against the VSSA client; map it to a canonical system only when the system is explicitly named or the scope is unambiguous. Record the portal organization ID, `authorityId`, `orgGroupId`, procedure/notice/document/resource identifiers, procurement stage/type, title, authority, relevant dates and status, direct stable page/document URL, supporting passage, access date, and confidence. Apply strict stage semantics: + +- an RFI or market consultation proves market engagement/planned scope only; +- an RFP, tender notice, or technical specification proves intended procurement only; +- a supplier offer proves a bid only; +- a voluntary ex-ante transparency notice may identify an intended or sole-source supplier and record a winning offer, but it does not by itself prove that a contract was signed; preserve the notice type, bounded supplier role, offer value and date, and keep contract date/delivery unknown unless the body or a separate signed-contract/award record establishes them; +- an award notice or signed contract is needed to establish selected supplier and contractual role; when an award names the winner but omits the contract-signature date, record the award and role while leaving the contract date unknown rather than substituting publication, dispatch, winner-selection, or offer dates; +- implementation, production use, hosting, and completion/acceptance each need separate direct evidence. + +Search-result rows and snippets are discovery leads, not citations. Open the notice and relevant attachments, classify each document, and cite the source body that supports the exact claim. Keep unresolved mappings explicit rather than forcing a procurement to a similarly named client or system. + +### 6. Verify and triangulate + +Open every source. Search snippets are discovery leads, never evidence. + +Apply the claim/source test: + +- Does the source name the exact client? +- Does it name the exact system or unmistakably define it? +- Does it state the relationship claimed? +- Does it identify the contractor’s role rather than merely mention the company? +- Does the date support current, historical, planned, or completed status? + +For contractor claims, seek two independent sources when practical: one buyer/public record and one supplier/project source. A single explicit official award or signed-contract record may justify high confidence. A vendor case study alone is normally medium unless detailed and independently corroborated. + +## Source Priority + +1. Signed contracts, award notices, official system regulations, technical specifications, acceptance/completion records. +2. Official client, ministry, procurement, legal-register, funding, or open-data pages. +3. Contractor case studies and official company announcements that explicitly name the client and work. +4. Reputable professional or trade publications quoting named parties. +5. Conference slides, staff profiles, code repositories, job advertisements, DNS/CDN/IP observations, and technology detectors — leads only unless corroborated. + +Preserve the direct document/page URL, source title, publisher, relevant date, and access date. Prefer stable HTML or official PDF URLs. If a source is mutable, record enough title/date context to rediscover it. + +When dossiers are published as generated pages, retain citations at claim level and also generate a de-duplicated `## References` index per client and per derived system. The index is a navigation aid, not a replacement for precise evidence placement. De-duplicate by resolved URL while preserving a useful source label and first-seen order; combine client-observation URLs with canonical system-analysis evidence for system pages. Validate exact reference-index coverage across all generated records. + +## Confidence and Status + +### Confidence + +- **High:** direct official source explicitly proves the client-system relationship or contractor role; dates and identities align. +- **Medium:** explicit credible source, such as a detailed vendor case study, but independent corroboration or current-status evidence is missing. +- **Low:** indirect evidence useful as a lead; do not present as confirmed fact. + +### Dossier status + +- **researched:** the query ladder and primary-source classes were checked; identified rows are supported and contractor searches were performed. +- **partial:** systems were found but meaning, hosting, contractor, role, dates, or primary evidence remain materially incomplete. +- **no-public-evidence:** a genuine documented search found no defensible named system/cloud relationship. + +Do not mark `researched` merely because one system was found. + +## Cloud Attribution Rules + +Separate these claims: + +- the client uses centralized IT services; +- a system is hosted by a named provider; +- the contractor uses a cloud internally; +- the public website sits behind a CDN; +- the production information system runs in AWS, Azure, GCP, a national cloud, or a private data centre. + +Only the last claim establishes the system’s cloud. DNS, IP ownership, TLS, CDN headers, JavaScript libraries, job ads, and generic cloud framework references are leads, not proof of production hosting. + +An official completion page may explicitly say that a production version was moved into a named institution’s **production environment**. Preserve that statement as direct evidence of the named environment provider/operator role, but do not translate it into `cloud`, `on-premises`, `hybrid`, data-centre ownership, or commercial hosting unless the source separately identifies the environment type. In canonical lifecycle analysis, keep hosting classification `unknown`, attach the quoted production-environment evidence, and explain exactly which classification detail remains absent. Reuse the contractor registry’s established exact legal entity label for the named institution so an English rendering does not create a duplicate contractor identity. + +When the same official page explicitly states that the client and partner **began implementing the named system/project** in a stated month or day, that can establish delivery kickoff at only that precision when the page makes clear that the project created or delivered the system. Do not treat a later modernization, migration, integration, or support project start as the original system kickoff. Likewise, acceptance for operation or opening to external users without a stated date does not establish a dated latest production release; retain the operational fact at observation scope while keeping the canonical release date unknown. + +## Cross-client System Registries + +When client dossiers feed a unique systems/services registry, preserve the client row as the evidence observation and generate the registry as a derived view. Do not treat identical generic wording as one shared system, do not merge by fuzzy similarity alone, and do not edit generated system dossiers as evidence. Maintain reviewed alias rules, scope generic portals/websites to the client, keep composite labels separate until evidence supports splitting, and use stable identity-derived IDs. + +When the registry maintains canonical per-system infrastructure and lifecycle analysis, research and record these separately from client-observation rows: + +- **Hosting environment:** production `cloud`, `on-premises`, or `hybrid`, plus the named provider/environment when directly evidenced. +- **System kickoff:** the original system delivery or implementation start, at only the date precision supported by the source. +- **Latest production release:** the latest directly evidenced production release or deployment date. + +Keep each field explicitly unknown when evidence is absent. Do not substitute DNS/IP/CDN signals, centralized-cloud eligibility, a supplier's generic cloud capabilities, procurement publication, contract signature, project completion, webpage update dates, or client-specific adoption facts. Every known value requires direct system-level evidence and should preserve a supporting quote. Shared-system claims must apply to the system as a whole; otherwise keep the fact at observation/client scope. + +A procurement or support notice can directly establish production cloud hosting even when it does not name the supplier or cloud provider, but only when its body explicitly says that the production system operates in the cloud infrastructure covered by the service (for example, rental and operation of the cloud infrastructure **in which the named chatbot runs**). Record `model: cloud`, describe the environment as unnamed, retain the provider as unidentified, quote the decisive wording, and do not infer AWS, Azure, GCP, ownership, region, tenancy, or supplier identity. Before attaching this evidence to canonical analysis, inspect every observation grouped under that system identity. If the label is generic—such as “intelligent chatbot”—scope it to the client first unless direct evidence proves all grouped observations are the same shared system. When scoping creates a new system ID, migrate only the client-specific evidence to the scoped ID and leave the former generic record unknown unless it has independent system-wide evidence. + +See [`references/cross-client-system-registry.md`](references/cross-client-system-registry.md) for the full identity, provenance, generation, validation, and continuous-research method. When contractor/provider observations need a first-class cross-client view, follow [`references/derived-contractor-registry.md`](references/derived-contractor-registry.md) for evidence-preserving entity extraction, stable `CTR-*` identities, navigation, test-first generation, and deployment validation. On generated client pages, keep every named contractor attached to the exact system observation where its role is evidenced and link the entity to its derived `CTR-*` contractor card; leave unknown placeholders explicit and unlinked. Validate complete Client→system→contractor-card link coverage both during generation and against rendered production routes. For canonical per-system cloud/on-premises, kickoff, and latest-release fields—including evidence objects, exact-coverage validation, identity migration, and automation rules—follow [`references/system-infrastructure-lifecycle-analysis.md`](references/system-infrastructure-lifecycle-analysis.md). + +## Batch Execution Discipline + +For recurring registry batches, protect delivery from unbounded research: + +1. Preflight repository access, recover exact target identities, and inspect the validator before researching. Parse the registry according to its actual record layout: numbered records may place the ID and client name on separate lines with blank lines between them. Validate the complete unique ID sequence and non-empty names from parsed record blocks; do not assume an `ID. name` one-line regex. Before creating files, build a target manifest containing `(ID, exact source name, canonical indexed path, current status)` by joining the authoritative source to the existing index. Compare every target tuple and use the index's established path when present; never reconstruct ID order or filenames from a task summary, adjacent organizations, or memory. Run the validator immediately after the first write batch so an ID/name/path transposition is caught before research is committed. +2. Audit coverage programmatically before selecting targets: count dossier statuses, canonical system rows, missing contractor cells, low-confidence rows, and generated unique-system/observation totals. Rotate to the next weak block rather than repeatedly selecting familiar records, but treat these counts as triage only—read each selected dossier and its direct sources before changing claims. +3. Work one client at a time through identity, purpose, systems, contractor ladder, dossier write, and citation check. Save each defensible dossier as soon as it is complete instead of postponing all writes until every search ends. When adding a generically named system, inspect the current cross-client grouping before editing canonical infrastructure/lifecycle analysis. If the new evidence is client-specific, scope the observation label first, bootstrap the new canonical analysis entry, move only the applicable evidence, and rerun generation before continuing; this prevents a true client-specific hosting fact from contaminating unrelated observations that happened to share the same generic label. +4. When dossier edits introduce new canonical system identities, run the repository's system-analysis bootstrap **before** the aggregate validation command if its test suite checks exact analysis-ID coverage before the validation script reaches its own bootstrap step. A failing aggregate validation that reports only missing analysis IDs is not evidence that the dossier rows are invalid: bootstrap once, rerun the full validator, and require the second run to pass. If an alias change merges identities, first inspect every stale analysis entry; delete only all-unknown entries, or migrate researched evidence to the surviving identity before bootstrapping. Reconcile the resulting system and observation count delta explicitly. +5. Time-box blocked search-provider retries and switch to official-domain routes, child sitemaps, internal search, legal/procurement sources, or a conservative `partial`/`no-public-evidence` dossier. Repeated discovery attempts are not more valuable than a validated artifact. +6. Reserve enough execution capacity for index updates, repository validation, diff review, commit, push, ref equality, and authenticated artifact readback. Do not spend the full tool or time budget on discovery and leave the requested deliverable unwritten. +7. Treat the channel wiki's latest-rotation block as durable history, not a replaceable status slot. Before writing a new latest block, demote the previous latest block intact under a dated `Previous scheduled rotation` heading; then insert the new block and verify that both rotations remain present. Record the next rotation only after delivery verification. This prevents a successful current run from silently erasing the immediately preceding run's IDs, evidence, counts, and deployment state. +8. If a source uncovers a similarly named institution, stop and rerun the identity-collision check before following its systems or contractors. +9. When converting legacy dossiers to the canonical six-column format, inventory the original system/service labels before rewriting and compare them with the rewritten rows afterward. If the underlying system identity has not changed, preserve the original system-label text byte-for-byte and improve only the new `Meaning`, relationship, contractor, and evidence cells; even a harmless-looking deletion such as `access`, `with VSSA`, or `Additional` changes a name-hashed `SYS-*` identity and can create missing/extra analysis IDs. Rename only when correcting a real identity error, and then perform the documented evidence migration/stale-entry review deliberately. Record the pre-migration generator counts, regenerate immediately after the first converted batch, and reconcile **any** unexpected count change—not only decreases. A count increase can be legitimate when the generator previously ignored legacy four-column tables and begins ingesting their rows after conversion, but verify that each new observation corresponds to an inventoried legacy row and that new system/contractor identities are not punctuation or legal-designator duplicates. A decrease can signal a silently discarded defensible row. Do not advance to the next batch until the delta is explained. + + Legacy tables sometimes encode one vendor-backed service as two rows: one row for the service and another whose “system” label is only the supplier or e-solution credit. In the canonical six-column table, merge these into one service observation and place the supplier plus bounded role in `Contractor / company`; do not preserve a company name as a separate system merely to keep counts stable. This legitimate normalization changes the canonical identity set. Before bootstrapping, inspect the stale `SYS-*` analysis entry: delete it only if all lifecycle/hosting fields are still unknown; if it contains researched evidence, manually migrate that evidence to the correct surviving/new system identity when semantically applicable. Then bootstrap missing analysis entries, regenerate, and verify the expected observation/system delta and contractor extraction. Never make a bootstrap helper silently delete stale analysis IDs. +10. On `*.lrv.lt` procurement pages, inspect the raw attachment links as well as visible HTML. Plans and reports are often exposed only as generically named `/media/`, `/public/canonical/`, PDF, DOC, or XLS/XLSX links, so an HTML keyword scan can miss the actual procurement contents. Open and classify relevant attachment bodies before using them; a plan still proves intent only, not award or delivery. +11. Treat a completed no-new-claim rotation as a valid conservative research outcome. If the assigned source classes were genuinely reopened, identity and contractor/system query ladders were rerun, and existing claims remain supported, advance each dossier's `Last checked` date and the matching index date while preserving statuses, rows, confidence, and unresolved gaps. Do not advance dates after only a superficial availability check, and do not promote a status merely because the existing rows were reconfirmed. In the delivery report, state which source classes were checked and explicitly distinguish “no defensible new claim” from “no search performed.” + +## Iterative Skill Improvement + +After each batch: + +1. Review which queries and source classes produced defensible new systems or contractors. +2. Record reusable Lithuanian terminology, portal locations, document types, and verification traps. +3. Patch this skill only with repeatable findings validated in real investigations; do not add client-specific facts. +4. Remove or correct stale URLs and ineffective or misleading instructions. +5. Keep the output template stable unless the user changes the desired dossier format. + +Examples of valid skill improvements: a newly confirmed procurement search route, a recurring legal-document phrase that reveals system processors, or a reliable method for locating vendor portfolios. Client-specific system names belong in dossiers, not in this skill. + +## Common Pitfalls + +1. Repeating the same evidence in a table, findings section, evidence list, and search log. +2. Treating eligibility for a shared service catalogue as proof that every catalogue service is consumed. +3. Naming a developer when the source only proves maintenance, licensing, or hosting. +4. Treating a tender as proof that the advertised contract was awarded or completed. +5. Presenting a historical contractor as the current supplier without dates. +6. Assuming that a public website’s host is the host of a protected back-office system. +7. Using vendor logos or unsourced client lists without a project description. +8. Stopping at the client’s site instead of searching supplier portfolios and procurement records. +9. Guessing when public evidence is absent. +10. Treating a VPT `downloadContractDocument` URL—or an institution-site attachment labelled `Pagrindinė sutartis`, `Preliminarioji sutartis`, or `Tiekėjo pateiktas pasiūlymas`—as proof of award, signature, supplier role, or delivery without opening and classifying the document body. A supplier offer proves a bid, and a preliminary agreement does not by itself prove a call-off. +11. Translating `asmens duomenų valdytoja` as system owner when the page proves only a personal-data controller role. +12. Patching prose inside a Markdown table without rechecking the column count and accidentally deleting a cell. For pipe tables, strip the leading and trailing empty segments before counting (`row.split('|')[1:-1]`); the required dossier table has exactly six cells. Counting the raw `split('|')` result causes false failures because it includes both boundary empties. +13. Normalizing punctuation in registry-backed names (for example, dropping Lithuanian quotation marks) instead of copying the authoritative name byte-for-byte into the heading and index. +14. Assuming an index-generation helper exists. Inspect repository scripts first; if no generator is present, update the index using its established row format and run the repository validator. +15. Assuming registry IDs and names occupy the same line. Some authoritative lists use numbered blocks with blank lines between the ID and name; a one-line regex can return every ID with an empty name while still appearing to find the full sequence. +16. Citing a login or application endpoint merely because it is the system URL. Before delivery, extract and check **every Markdown URL in each changed dossier**, including the `Purpose` sentence as well as table evidence; do not validate only the system rows. Use a browser-equivalent `GET`, follow redirects, and inspect enough of the returned page to confirm the cited claim—not merely an HTTP status or the first bytes. If an operational endpoint or purpose link returns an error, retain it only when necessary and pair or replace it with an opened official description, legal document, or government service-catalogue record that actually proves the claim. Some hosted appointment/login endpoints return an automation-specific status such as `417` while the institution's accessible official page still links the exact endpoint; in that case, record the endpoint result, use the opened institution page as the relationship evidence, and keep provider attribution bounded to what the endpoint identity and provider terms actually establish. Never create a blanket exception for that status or domain. Do not treat endpoint availability itself as evidence of ownership, supplier, or current operation. +17. Treating a vendor page's bot-sensitive `403` as a broken citation without a browser-equivalent retry. Some public Lithuanian supplier case-study pages reject a generic Python or curl user agent but return the full page to a current browser user agent plus normal HTML `Accept` and language headers. Retry once with browser-equivalent request headers, open and inspect the returned body, and record the status; if it still fails, replace or corroborate the citation rather than weakening verification or bypassing authentication controls. +18. Naming a feasibility-study or technical-specification supplier as the system developer. Preparatory contracts prove only the scoped analysis/specification role unless a later award, delivery record, or explicit project source establishes implementation; state the date and preparatory role in the contractor cell. +19. Collapsing repeated generic labels into one cross-client system. “Official website”, “institution portal”, and similar phrases may describe different client-specific resources; scope them to the client unless direct evidence proves a shared identity. Conversely, preserve reviewed aliases for clearly identical named products instead of producing duplicate registry entries. +20. Introducing a new contractor identity by casually changing legal-designator or consortium wording in one observation. Before writing a contractor cell, inspect the generated contractor registry and existing canonical observations for that supplier. Reuse the established exact entity label when it denotes the same evidenced entity; keep a consortium phrase separate from the standalone company, and do not prepend `UAB` to an established `„Company“ su konsorciumo partneriais` label unless evidence and reviewed identity rules justify changing that composite identity. After generation, compare contractor counts and inspect all near-name matches; an unexpected increase commonly signals accidental fragmentation rather than a new contractor. +21. Flattening an award's winner, participant-group leader and subcontractor into one generic contractor attribution. Read the award body's official winner name, any `Pirkimų procedūros dalyvio vadovas` / participant-group leader label, and subcontracting fields separately. Preserve the official selected supplier as the winner; treat a separately displayed participant-group leader only as portal metadata unless the body explicitly establishes a distinct legal entity and role; preserve an explicitly named subcontractor with its stated share/value and bounded subcontractor role. Do not concatenate these labels into one contractor identity or infer consortium membership, co-award, implementation scope, production use, or hosting. See [`references/cvp-is-organization-route.md`](references/cvp-is-organization-route.md) for the award extraction checklist. + +## Verification Checklist + +- [ ] One concise file exists for the client. +- [ ] Exact official name and one-sentence purpose are present; for registry-backed batches, heading and index names match the authoritative source exactly, including punctuation. +- [ ] Every row explains what the system does. +- [ ] Client relationship is precise. +- [ ] Contractor/company and role are named when publicly evidenced. +- [ ] Contractor searches were performed for every named system. +- [ ] Historical/current/planned status is not blurred. +- [ ] Every material claim has a direct opened source. +- [ ] CPVA search was attempted with exact client/system variants where relevant, and only opened source bodies—not result snippets—were used as evidence. +- [ ] Generated client and system pages retain claim-level citations and include de-duplicated References indexes. +- [ ] Cloud attribution is corroborated, not inferred from infrastructure hints. +- [ ] Gaps are explicit and short. +- [ ] The skill was reviewed for validated reusable improvements after the batch. diff --git a/references/cross-client-system-registry.md b/references/cross-client-system-registry.md new file mode 100644 index 0000000..c7d9e0e --- /dev/null +++ b/references/cross-client-system-registry.md @@ -0,0 +1,45 @@ +# Cross-client system identity and registry method + +Apply this after canonical client dossiers contain structured system rows. + +## Observation first + +Treat each client/system row as an observation, not automatically as a globally unique system. Preserve: + +- exact observed label; +- client ID and exact client name; +- relationship and plain-language meaning; +- contractor/company with bounded role; +- direct evidence and confidence; +- canonical dossier path. + +A generated registry must be traceable back to every observation and must not become a second editable evidence store. + +## Identity decisions + +Use conservative tiers: + +1. **Exact normalized named identity** — case/whitespace/dash/Markdown differences may aggregate when the label is clearly a named product, domain, acronym, or shared service. +2. **Reviewed alias identity** — merge variant labels only when an official source, unmistakable product identity, or explicit maintained rule proves equivalence. +3. **Client-scoped generic identity** — identical phrases such as official website, institution portal, virtual exhibition, or public-service site remain separate per client. +4. **Composite/ambiguous identity** — labels joining multiple systems remain separate until research can split them without losing the stated client relationship. + +Normalization is a matching aid, not evidence. Never use fuzzy similarity alone to merge records. Preserve observed labels as aliases and explain the registry classification. + +## Stable generated dossiers + +- Prefer hash-derived IDs/slugs from the durable identity key rather than alphabetic sequence numbers. +- Include observed aliases, distinct client count, observation count, confidence distribution, and a client-relationship table. +- Carry direct evidence through from canonical rows. +- For legacy rows without separate meaning/contractor columns, label those fields as not separately structured rather than inventing values. +- Emit machine-readable JSON and validate unique IDs/slugs, observation totals, client references, and one route per unique record. + +## Continuous research loop + +During each recurring research batch: + +1. Improve canonical client rows with exact official system names, purpose, relationship, contractor role, date, evidence, and confidence. +2. Review new labels against the alias rules. +3. Record a merge only when equivalence is defensible; otherwise preserve separation. +4. Regenerate and compare unique-system and observation counts. +5. Investigate high-frequency ambiguous labels and missing contractors as priority research targets. diff --git a/references/cvp-is-organization-route.md b/references/cvp-is-organization-route.md new file mode 100644 index 0000000..f366522 --- /dev/null +++ b/references/cvp-is-organization-route.md @@ -0,0 +1,62 @@ +# CVP IS organization-first procurement discovery + +Use this route as an additional discovery path for Lithuanian public-procurement evidence. It is not a complete procurement history. + +## Validated route + +1. Open `https://viesiejipirkimai.lt/epps/viewOrganisations.do` and search the exact client legal name, then former-name and abbreviation variants where needed. The live organization form uses `POST /epps/viewOrganisations.do` with lowercase field names, including `within=template.group.ca`, `name=`, and optional `shortName`, `city`, and `street`; preserve the session cookie and normal referrer when reproducing the form request. Do not invent a GET query parameter or capitalize `name` from the input element's `id="Name"`. If a byte-exact authoritative name yields no row, retry only bounded typography/legal-form variants such as removing Lithuanian quotation marks or using the portal's likely `AB`, `VšĮ`, or expanded legal-form rendering. Record which query matched. +2. Identity-check the returned legal name and available profile details. Parse complete organization-result table rows and bind each displayed legal name to the `prepareViewCAOrganisation.do?id=...` link in that same row. Do not collect every profile link from a wide HTML context and then substring-match the query: one result page or surrounding fragment can contain multiple organization/profile links, producing a false “exact” match or an ambiguous pair. Preserve the query variant, portal-displayed organization name, registration number/domain when available, and organization ID. If an exact-name query still yields multiple distinct exact rows, keep the identity unresolved until registration number, domain, or another authoritative identifier distinguishes them; never select the first result. A variant match is discovery only until reconciled with the authoritative client identity. Treat near-identical domains as a collision risk: open the homepage and corroborate its institution name/contact domain against the CVP IS profile before using any system link found there. +3. Open `https://viesiejipirkimai.lt/epps/prepareViewCAOrganisation.do?id=`. +4. Extract the real target of **PERŽIŪRĖTI VISUS PASKELBTUS SKELBIMUS**. It has the shape: + `https://viesiejipirkimai.lt/epps/notices/viewPublishedNotices.do?authorityId=&orgGroupId=` +5. Read both IDs from that link. Never assume they match or derive one arithmetically. +6. Review all result pages, preferably at the largest supported page size. The live list exposes pagination through a query key shaped `d--p=`; derive the exact key and last-page value from the returned page's own navigation or `writePageSelection(...)` call rather than hard-coding the observed table ID. Preserve title, notice type/stage, status, language, upload date, and publication date. +7. Parse each HTML table row as a record. The notice **type** is commonly an anchor in the first cell, while the procurement **title** is plain text in the second cell—not an anchor. Therefore, anchor-text extraction alone can falsely report zero IT candidates even when a relevant title is visible. Extract and normalize all cells, keep the first-cell attachment URL, then keyword-filter the second-cell title. +8. Open promising notice-type links. They can be forced PDF attachments rather than HTML pages. + +## Direct PDF behavior + +A notice link commonly has this form: + +`viewPublishedContractNotice.do?resourceId=&documentId=¬iceType=&extId=&lang=lt¬iceId=&isNational=false` + +Browser navigation may report `ERR_ABORTED` because the response is a download. This is not evidence of failure. Fetch the exact URL directly and verify: + +- HTTP success; +- `Content-Type: application/pdf`; +- non-trivial size; +- `%PDF-` header; +- extractable text, or OCR when it is scanned. + +Preserve organization ID, `authorityId`, `orgGroupId`, `resourceId`, `documentId`, `noticeId`, `noticeType`, `extId`, procedure identifier, exact title/authority, dates, status, direct URL, supporting quotation, and access date. + +## Evidence semantics + +Classify the PDF from its body, not from the filename or result snippet: + +- market consultation/RFI → planned scope and market engagement only; +- tender/RFP/technical specification → intended procurement only; +- supplier offer → bid only; +- award notice or signed contract → selected supplier and bounded contractual role; +- amendment → only the change stated; +- implementation, production use, hosting, and acceptance/completion each require their own direct evidence. + +An award that covers multiple explicitly named information systems can support one observation row per materially distinct system even when the systems share one lot and supplier. Keep the shared contract value and role clearly scoped to the joint award—do not imply that the full value belongs independently to every row. Bootstrap unknown canonical lifecycle-analysis entries for each new stable system identity before running strict registry validation; then keep hosting, original kickoff, and latest production release unknown unless separately evidenced. + +For an award PDF, extract the exact buyer and system title, procedure identifier, winner, contract value, winner-selection date, contract-signature date, duration when stated, and subcontracting fields. If the body explicitly names a subcontractor together with its share or value, record that entity separately as a **subcontractor**, not as a co-winner, generic contractor, or inferred implementation partner. Preserve the award's own legal-entity spelling in the evidence record, then reconcile it conservatively with the existing contractor registry before creating a new identity. Do not treat an award's signature date or service duration as system kickoff, production release, acceptance, completion, or hosting evidence. + +Map a notice to a canonical system only when it explicitly names the system/acronym or the scope is otherwise unambiguous. Similar IT terminology is insufficient. + +## Confirmed examples and pitfalls + +- A validated exact-name search can identify an organization even when external search is poor: organization `3350` exposed `authorityId=3350` and distinct `orgGroupId=3432`. This reinforces that both IDs must come from the profile link, not arithmetic or equality assumptions. +- An exact organization match and an opened notice list can legitimately yield no relevant IT procurement. Record the checked IDs and negative outcome for rotation continuity, but do not add a dossier row or claim procurement coverage; the organization route is incomplete. +- When rotating several organizations, preserve each exact `organizationId` and the profile-derived `(authorityId, orgGroupId)` pair in the durable rotation log—even when the notice scan yields no claim. This gives the next run a reproducible identity checkpoint without converting discovery metadata into dossier evidence. +- If an exact authoritative organization name returns no byte-exact result, state that explicitly and do not force a near-name profile. Continue bounded legal-form/former-name checks and direct system/procurement routes; absence from the organization search is not evidence that the organization has no procurement history. +- Organization/profile IDs and group IDs can differ (for example, `authorityId=1297`, `orgGroupId=1298`). +- Organization notice lists may contain hundreds of records and include unrelated purchases; scan exact system names/acronyms plus IT terms, then open candidates. +- One organization can expose separate canonical systems in adjacent notices. Do not merge them merely because the same authority procured both. +- The organization list may omit predecessor organizations, parent-authority purchases, older migrations, contracts, amendments, or other procurement stages. Continue direct procurement searches and former-name checks. +- Award and tender notices with the same title are distinct evidence objects. Open both; do not infer the award supplier from the tender. +- A directly extracted award PDF can bind a public-facing portal label to a formal backend name and bounded supplier role. Preserve both labels rather than silently renaming the client observation. For example, the title may say only “Informacinio portalo palaikymas ir vystymas” while the body explicitly scopes the work to a named information system's portal, states the contract duration, winner, value, and signature date. Use the client page to retain the public service label and the award body to state the formal system scope and maintenance/development role. This still does not prove production hosting, original development, or release dates. +- When a command-line PDF extractor is unavailable, use an isolated dependency invocation such as `uv run --with pymupdf python ...` and verify the downloaded file begins with `%PDF-`, has non-trivial size, and yields text before relying on it. This is a portable extraction fallback, not a reason to weaken document classification. diff --git a/references/derived-contractor-registry.md b/references/derived-contractor-registry.md new file mode 100644 index 0000000..1cf48c9 --- /dev/null +++ b/references/derived-contractor-registry.md @@ -0,0 +1,46 @@ +# Derived contractor registry method + +Use this when canonical client-system observations contain contractor/provider text and the documentation needs a cross-client contractor view. + +## Evidence model + +Treat the contractor registry as a **derived view**, never as a second editable evidence store. Every relationship must retain: + +- exact canonical client and system identity; +- explicitly named contractor/provider entity; +- bounded role as stated by the source (developer, implementer, service provider, licensor, host, support supplier, consortium member, public-sector operator, etc.); +- direct evidence URL and confidence; +- historical/current/planned context where known. + +Do not turn `Not publicly identified`, “suppliers not named”, descriptive prose, or an ambiguous composite phrase into entities. Do not silently upgrade one role into another: development does not prove hosting or current support. + +## Identity and stable routes + +Normalize only safe textual differences for matching. Preserve the displayed source name and keep ambiguous composites separate until evidence supports a split or merge. Generate stable `CTR-*` IDs and slugs from a durable normalized identity key rather than sequence position. Sequence numbers may order generated files but must not define identity. + +Before adding or changing a contractor label in a canonical observation, inspect the current derived registry and all existing near-name observations. Reuse the established exact label only when it denotes the same evidenced entity. Treat a standalone company and a composite such as `„Company“ su konsorciumo partneriais` as separate identities unless reviewed evidence supports a merge. Legal-designator changes (`UAB`, `AB`, quotation style) can create a new deterministic identity; do not introduce them casually. After regeneration, compare the entity count and inspect near-name groups. An unexpected increase is an identity-collision warning, not evidence that a new contractor was discovered. + +A contractor dossier should list exact system/client relationships, bounded roles, confidence, and direct sources. A system dossier should expose its evidenced contractors, but the underlying client observation remains canonical. + +## Implementation sequence + +1. Write focused failing tests for named-entity extraction, unknown-placeholder exclusion, multi-contractor splitting, stable IDs/slugs, relationship aggregation, and one representative evidence-backed system. +2. Implement extraction and aggregation without editing generated pages directly. +3. Generate contractor overview/registry, individual dossiers, and machine-readable JSON. +4. Add a separate documentation collection with first-level **Overview** and expanded **Registry**, plus navbar/footer/search integration. +5. Add build invariants: JSON exists, IDs/slugs are unique, route count matches entity count, every relationship resolves to valid client/system IDs, and representative evidence is present. +6. Run focused tests, registry validation, type checking, full production build, and rendered browser checks for both a system page and contractor page. +7. After publishing, verify local/remote ref equality, exact CI run success, authenticated source readback, and deployed route behavior. An expected authentication redirect proves the access boundary is active, not page content; use successful CI publication and an authorized/backend content check when available before claiming deployed content. + +## Lifecycle evidence interaction + +A dated official announcement whose title/body explicitly says a named system started operating, launched, or went live can establish a production-release date at the source’s stated precision. This is different from generic page publication/update metadata. Quote the operational statement, cite corroborating announcements, and use the earliest date that explicitly establishes operation when official announcements differ. Do not infer delivery kickoff from go-live; kickoff needs separate evidence. + +## Common pitfalls + +- Parsing every capitalized phrase or semicolon clause as a company. +- Merging legal-name variants or consortium descriptions without reviewed identity evidence. +- Showing a contractor on a system page without retaining the source relationship. +- Presenting a historical creation supplier as the current maintainer. +- Claiming deployment verification solely from an unauthenticated redirect to an identity provider. +- Adding navigation without search indexing, generated-route validation, or machine-readable output. diff --git a/references/lithuanian-source-routes.md b/references/lithuanian-source-routes.md new file mode 100644 index 0000000..fb789e5 --- /dev/null +++ b/references/lithuanian-source-routes.md @@ -0,0 +1,57 @@ +# Lithuanian source-route notes + +Use these routes as discovery and verification aids; they do not lower the claim/source standard in the main skill. + +## Official institution sites + +- On LRV-hosted and similar official sites, inspect `/sitemap.xml` to enumerate deep pages that menus and search engines miss. Prioritize paths for `asmens-duomenu-apsauga`, `informacines-sistemos`, `registrai`, `projektai`, `viesieji-pirkimai`, `nuostatai`, and `atviri-duomenys`. +- Open the discovered page and cite the canonical page URL. A sitemap entry proves that a page exists, not the claim inside it. +- Pages headed `... asmens duomenų valdytoja` or `... asmens duomenų tvarkytoja` can directly establish a controller/processor role for listed systems. Do **not** translate `valdytoja` on a personal-data page into system ownership unless system regulations or another source explicitly assign ownership. + +## Official-site internal search fallback + +- When external engines do not expose an older institution or article, try the official municipality/institution search route directly. A recurring Lithuanian pattern is `/search?q=`. +- Treat the search page only as discovery. Open and cite the resulting official article; verify that the article body, not merely a search snippet, names the institution and supports the claimed purpose or relationship. +- This route is particularly useful for schools with retired domains, renamed institutions, and municipality-hosted news archives. + +## data.gov.lt fallback for blocked institution sites + +- When an institution's official site remains blocked after one browser-equivalent retry, search `https://data.gov.lt/datasets/?q=`. Inspect the organization/creator facets and result cards, then open the direct organization page (`/orgs//`) and each relevant dataset record (`/datasets//`). +- The organization page can corroborate the exact publisher identity, supervising jurisdiction, and sector. A direct dataset record can establish that the institution publishes or maintains data from a named register or system when its metadata says so. +- Treat the query page and facets as discovery, not final evidence. Cite the opened organization or dataset record. Dataset publication does not by itself prove software ownership, development, maintenance, hosting, or cloud provider; preserve those as unknown unless another source assigns the role. + +## CPVA search form + +- CPVA's `https://cpva.lt/paieska` uses Search & Filter Pro. The live, validated result URL is `https://cpva.lt/paieska?_sf_s=`. The browser rewrites the form state to this URL and the server-rendered `.search-filter-results-3173` container contains the result cards. An ordinary `POST` with `_sf_search[]` can return an apparently successful but unfiltered/empty page; do not use that as evidence that a query ran. +- Validate automation against a positive control such as the generic term `sistema` and a negative/random control. Confirm the result container changes, extract result-card titles and links, and recognize the explicit `Rezultatų nėra – įveskite / pakoreguokite paieškos frazę` empty state. +- CPVA search behaves as broad token matching rather than reliable exact-phrase matching. Long system labels can return unrelated pages sharing generic words, and even every queried system may appear to have a raw hit. Treat candidate counts as noisy discovery output; rank pages by exact acronym/name occurrence and open the source body before mapping anything. +- Search pages remain discovery evidence only. Open the resulting CPVA article, project, or attachment and require its body to prove the exact client/system relationship, project, date, funding, supplier, or delivery role before citing it. + +## Official policy PDFs as SaaS evidence + +- Inspect sitemap URLs for direct PDFs as well as HTML pages. School and care-institution sitemaps often expose electronic-diary usage rules, data-processing policies, or director-approved procedures that navigation menus omit. +- Extract and inspect the full PDF text. A dated official procedure can directly name the institution, product, provider, system purpose, administrative roles, and login URL; embedded PDF link annotations may also reveal a product URL not obvious in extracted prose. +- Bound the date carefully: a policy proves use or an assigned provider role at the policy date, not an unchanged current subscription, current maintenance allocation, contract term, or production hosting. Keep those fields unresolved unless a current contract or equivalent source establishes them. +- When the official policy names both the SaaS product and company, it can support the client-product relationship and provider attribution in one primary source. Open the provider's current legal/product page separately to corroborate the company's identity and product role, but do not inflate that corroboration into implementation, hosting, or contract claims. + +## Lithuanian public-procurement documents + +- Search results may expose direct documents at URLs shaped like `https://viesiejipirkimai.lt/epps/cft/downloadContractDocument.do?resourceId=...&documentId=...`. +- Official institution sites may also place a child `/sutartys/` page under a named system or service page. Inspect its raw anchors: these catalogues can separately link a preliminary agreement, amendment, main contract, and supplier offer even when the visible page contains almost no supplier detail. Resolve relative attachment URLs against the page URL, download each relevant body, and classify each document independently. +- Open and inspect the document body. Endpoint names and link labels such as `downloadContractDocument`, `Pagrindinė sutartis`, or `Tiekėjo pateiktas pasiūlymas` describe the expected document class but are not proof of supplier identity, signature, dates, or delivery until the body confirms them. +- If an HTML catalogue opens but an attachment rejects a generic downloader, retry once with the same browser-equivalent `User-Agent`, `Accept`, and language headers used for bot-sensitive public pages. If the body still cannot be opened, cite the catalogue only for the existence of procurement documents and leave supplier attribution unresolved. +- Classify evidence by its contents: + - technical specification or tender conditions → intended procurement and required scope; + - supplier offer → bidder and proposed scope, not award by itself; + - preliminary/framework agreement → framework parties and scope, not necessarily a call-off; + - main/signed contract → contracted role and dates; + - award notice → selected supplier, subject to identity/date alignment; + - acceptance/completion record → delivered or accepted work. +- Technical specifications are especially useful for current-system inventories, acronyms, integrations, migration scope, and planned functionality, but they do not identify the winning supplier unless the body explicitly does so. + +## Pre-delivery link verification + +- Extract every URL from changed dossiers and open each one before commit. Correct path truncation, stale slugs, and redirects before assigning confidence. +- Inspect redirect chains, not only the initial URL's status. An official institution page may redirect to a retired or unreachable product domain: the redirect can still prove that the institution links or historically linked the named service, but it does **not** prove that the service is currently operational. State the availability gap explicitly and avoid an unqualified current-use claim. +- For a multi-file batch, check unique evidence URLs concurrently with bounded timeouts, then retry flagged links once with a different network strategy (for example IPv4-only `curl -4`) before classifying them. Preserve genuine HTTP failures and unreachable final destinations as review items rather than silently accepting the source URL. +- Re-run the dossier table/schema check after any textual patch; a targeted replacement can accidentally remove a Markdown table cell while leaving the prose readable. diff --git a/references/system-infrastructure-lifecycle-analysis.md b/references/system-infrastructure-lifecycle-analysis.md new file mode 100644 index 0000000..40ec39c --- /dev/null +++ b/references/system-infrastructure-lifecycle-analysis.md @@ -0,0 +1,135 @@ +# System infrastructure and lifecycle analysis + +Use this method when a cross-client registry needs per-system answers for hosting location, system kickoff, and latest production release. + +## Scope boundary + +Keep these facts in canonical **system-level metadata**, separate from client-observation rows: + +- Client rows prove that one organization owns, operates, uses, procures, or depends on a system. +- System metadata describes the unique system as a whole. +- A client's adoption date, tenant hosting arrangement, or modernization project must not become a shared-system fact unless the source explicitly applies it to the system globally. + +Generated system dossiers remain derived views and must never become editable evidence stores. + +## Required fields + +Every unique system record should contain all three fields, even when unknown: + +1. `hosting_environment` + - known models: `cloud`, `on-premises`, `hybrid` + - record the named provider or environment when the source states it +2. `kickoff_date` + - original system delivery or implementation start + - precision: `year`, `month`, or `day` +3. `latest_release_date` + - latest directly evidenced production release or deployment + - precision: `year`, `month`, or `day` + +Allowed status values are `known`, `unknown`, and `not-applicable`. Unknown and not-applicable values require a concise reason. Known values require direct evidence. + +## Evidence object + +For each known value, preserve at least: + +```json +{ + "url": "https://official.example/source", + "title": "Official source title", + "quote": "Exact text supporting the asserted value" +} +``` + +The quote is a claim guard: it makes reviewers check whether the source really says *production hosting*, *kickoff*, or *release*, rather than merely mentioning a nearby date or technology. + +## Meaning of the fields + +### Hosting environment + +Accept only explicit production placement evidence. A source may identify a commercial cloud, government cloud, private cloud, institutional data centre, supplier data centre, on-premises deployment, or hybrid arrangement. + +Do not infer hosting from: + +- DNS, IP ownership, TLS, CDN, or public website hosting; +- eligibility for centralized IT/cloud services; +- a supplier's generic AWS/Azure/GCP capability; +- a SaaS product name without an institution/system-specific hosting statement; +- development, test, backup, or disaster-recovery infrastructure when production is not identified. + +### System kickoff + +Use the original system-delivery or implementation start only when the source describes it as commencement, start, kickoff, or equivalent. Do not silently substitute: + +- tender publication; +- contract signature; +- funding approval; +- public launch; +- modernization start; +- project completion. + +If only a later modernization kickoff is known, keep the original system kickoff unknown and preserve the modernization date in its proper project/observation context. + +### Latest production release + +Use an explicit production release, deployment, go-live, or version release date. A dated official announcement whose title/body explicitly says the system **started operating**, **launched**, **went live**, or was **deployed to production** may use the announcement's stated publication date as the release date: the operational statement establishes the event and the official date anchors it. Prefer the earliest authoritative announcement when corroborating reposts differ by a day, preserve all corroborating sources, and do not treat a later repost as a later release. + +Do not substitute: + +- a webpage updated or publication date when the body does not explicitly establish production operation; +- contract end or project completion; +- acceptance date unless the source also proves production deployment; +- latest tender or maintenance contract date. + +## Identity-safe canonical storage + +Prefer a canonical metadata file keyed by the registry's stable system ID and include the underlying identity key as a guard: + +```json +{ + "schema_version": 1, + "systems": { + "SYS-XXXXXXXXXX": { + "identity_key": "name:canonical identity", + "hosting_environment": {"status": "unknown", "reason": "...", "evidence": []}, + "kickoff_date": {"status": "unknown", "reason": "...", "evidence": []}, + "latest_release_date": {"status": "unknown", "reason": "...", "evidence": []} + } + } +} +``` + +Require exact coverage of all generated system IDs. Missing records must fail validation rather than silently defaulting to unknown, because omission and researched-but-unknown are different states. + +When aliases, normalization, or client-scoped labels change, generated IDs may change. A bootstrap helper may add missing unknown skeletons, but it must: + +- never overwrite researched values; +- stop on stale IDs; +- require manual evidence migration for merges and splits; +- preserve the `identity_key` check. + +## Validation rules + +Fail generation/build when: + +- metadata IDs differ from generated registry IDs; +- an identity key does not match; +- any required field is absent; +- a status or hosting model is outside its controlled vocabulary; +- a known value lacks a direct HTTP(S) source, title, or quote; +- an unknown/not-applicable value lacks a reason; +- a date does not match its declared precision; +- generated JSON omits the analysis fields or retains an old schema version. + +Publish canonical metadata as a byte-identical raw artifact with a checksum. Include the analysis in machine-readable registry JSON and generated system pages. + +## Research rotation + +A recurring research job should, for every unique system touched in a client batch: + +1. Search explicitly for production hosting, original kickoff, and latest production release. +2. Open and classify every candidate source. +3. Update only directly supported fields. +4. Preserve explicit unknowns for the rest. +5. Regenerate, validate, build, inspect the complete diff, and verify published page and JSON output. + +Initial migration should create explicit unknown records for full coverage. It must not mine existing prose automatically and turn nearby cloud or date mentions into system-level facts. diff --git a/scripts/validate_skill.py b/scripts/validate_skill.py new file mode 100644 index 0000000..f4c2235 --- /dev/null +++ b/scripts/validate_skill.py @@ -0,0 +1,22 @@ +#!/usr/bin/env python3 +from pathlib import Path +import re + +ROOT = Path(__file__).resolve().parents[1] +skill = ROOT / "SKILL.md" +text = skill.read_text(encoding="utf-8") +assert text.startswith("---\n"), "frontmatter must start at byte zero" +match = re.match(r"---\n(.*?)\n---\n(.+)", text, re.S) +assert match, "invalid frontmatter/body framing" +frontmatter, body = match.groups() +assert re.search(r"^name:\s*vssa-clients\s*$", frontmatter, re.M), "wrong skill name" +description = re.search(r'^description:\s*["\']?(.*?)["\']?\s*$', frontmatter, re.M) +assert description and 0 < len(description.group(1)) <= 1024, "invalid description" +assert len(text) <= 100_000, "SKILL.md exceeds Hermes limit" +assert body.strip(), "skill body is empty" +links = re.findall(r"\[[^]]+\]\((references/[^)]+)\)", text) +assert links, "no linked references" +unique_links = sorted(set(links)) +missing = [link for link in unique_links if not (ROOT / link).is_file()] +assert not missing, f"missing linked references: {missing}" +print(f"OK skill=vssa-clients chars={len(text)} references={len(unique_links)}")