1"use strict";(self.webpackChunkopenrefine_documentation=self.webpackChunkopenrefine_documentation||[]).push([[4054],{58497(e,n,t){t.r(n),t.d(n,{assets:()=>l,contentTitle:()=>r,default:()=>h,frontMatter:()=>i,metadata:()=>a,toc:()=>c});const a=JSON.parse('{"id":"manual/wikibase/advanced-schemas","title":"advanced-schemas","description":"Sometimes your data is not as simple as a normal table, or the sort of","source":"@site/docs/manual/wikibase/advanced-schemas.md","sourceDirName":"manual/wikibase","slug":"/manual/wikibase/advanced-schemas","permalink":"/docs/manual/wikibase/advanced-schemas","draft":false,"unlisted":false,"editUrl":"https://github.com/OpenRefine/openrefine.github.com/edit/master/docs/manual/wikibase/advanced-schemas.md","tags":[],"version":"current","lastUpdatedBy":"Antonin Delpeuch","lastUpdatedAt":1672335305000,"frontMatter":{}}');var o=t(74848),s=t(28453);const i={},r=void 0,l={},c=[{value:"Hierarchical data",id:"hierarchical-data",level:2},{value:"Conditional additions",id:"conditional-additions",level:2},{value:"Varying properties",id:"varying-properties",level:2},{value:"Adapting to existing data on Wikibase",id:"adapting-to-existing-data-on-wikibase",level:2}];function d(e){const n={a:"a",code:"code",em:"em",h2:"h2",li:"li",p:"p",strong:"strong",ul:"ul",...(0,s.R)(),...e.components};return(0,o.jsxs)(o.Fragment,{children:[(0,o.jsx)(n.p,{children:"Sometimes your data is not as simple as a normal table, or the sort of\nstatements that you want to do varies on each row. This document\nexplains how to work around these cases."}),"\n",(0,o.jsx)(n.h2,{id:"hierarchical-data",children:"Hierarchical data"}),"\n",(0,o.jsxs)(n.p,{children:["Sometimes your source provides data in a structured format, such as XML,\nJSON or RDF. OpenRefine can import these files and will convert them to\ntables. These tables will reflect some of the hierarchy in the file by\nmeans of null cells, using the ",(0,o.jsx)(n.a,{href:"/docs/manual/exploring#rows-vs-records",children:"records mode"}),"."]}),"\n",(0,o.jsxs)(n.p,{children:["The Wikibase extension always works in rows mode, so if we want to add\nstatements which reference both the artist and the song, we need to fill\nthe null cells with the corresponding artist. You can do this with the\n",(0,o.jsx)(n.strong,{children:"Fill down"})," operation (in the ",(0,o.jsx)(n.strong,{children:"Edit cells"})," menu for this column).\nThis function will copy not just cell values but also reconciliation\nresults."]}),"\n",(0,o.jsx)(n.h2,{id:"conditional-additions",children:"Conditional additions"}),"\n",(0,o.jsx)(n.p,{children:"Sometimes you want to add a statement only in some conditions."}),"\n",(0,o.jsx)(n.p,{children:"The workflow to achieve this looks like this:"}),"\n",(0,o.jsxs)(n.ul,{children:["\n",(0,o.jsx)(n.li,{children:"Use facets to select the rows where you do not want to add any\ninformation;"}),"\n",(0,o.jsx)(n.li,{children:"Blank out the cells in the column that contain the information you\nwant to add. If you do not want to lose this information, you can\ncreate a copy of the column beforehand;"}),"\n",(0,o.jsx)(n.li,{children:"Remove your facets to see all rows again;"}),"\n",(0,o.jsx)(n.li,{children:"Create a schema using the column you partially blanked out as\nstatement value."}),"\n"]}),"\n",(0,o.jsx)(n.h2,{id:"varying-properties",children:"Varying properties"}),"\n",(0,o.jsx)(n.p,{children:"Sometimes you wish you could use column variables for properties in your\nschema. It is currently not possible, first because we do not have a\nreconciliation service for properties yet, but also because allowing\nvarying properties in a statement would mean that these properties could\npotentially have different datatypes, which would break the structure of\nthe schema."}),"\n",(0,o.jsxs)(n.p,{children:["If you only want to use a few properties, there is a way to go around\nthis problem. For instance, say you have a first column of altitudes and a\nsecond column that indicates whether you should add it as\n",(0,o.jsx)(n.a,{href:"https://www.wikidata.org/wiki/Property:P2254",children:"operating altitude (P2254)"})," or as\n",(0,o.jsx)(n.a,{href:"https://www.wikidata.org/wiki/Pro
1perty:P2044",children:"elevation above sea level (P2044)"}),"."]}),"\n",(0,o.jsxs)(n.p,{children:["Create a text facet on the first column. Filter to keep only the\n",(0,o.jsx)(n.em,{children:"altitude"})," values. Add a new column based on the second column, by\nkeeping the default expression (",(0,o.jsx)(n.code,{children:"value"}),") which just copies the existing\nvalues. Then, select the ",(0,o.jsx)(n.em,{children:"maximum operating altitude"})," value in the facet\nand do the same. Reset the facet, you should have obtained two new columns\nwhich partition the original column. You can now create a schema which adds\ntwo statements, with values taken from those columns. Since blank values are\nignored, exactly one statement will be added for each item, with the desired property."]}),"\n",(0,o.jsx)(n.h2,{id:"adapting-to-existing-data-on-wikibase",children:"Adapting to existing data on Wikibase"}),"\n",(0,o.jsx)(n.p,{children:"Sometimes you want to create statements only if there are no such\nstatements on the item yet. Here is one way to achieve this:"}),"\n",(0,o.jsxs)(n.ul,{children:["\n",(0,o.jsxs)(n.li,{children:["first, retrieve the existing values from Wikidata first, using the\n",(0,o.jsx)(n.strong,{children:"Edit columns"})," \u2192 ",(0,o.jsx)(n.strong,{children:"Add columns from reconciled values"})," action;"]}),"\n",(0,o.jsxs)(n.li,{children:["second, create a ",(0,o.jsx)(n.em,{children:"facet by null"})," on the newly created column that\ncontains the information you want to control against;"]}),"\n",(0,o.jsxs)(n.li,{children:["select the non-null rows (value ",(0,o.jsx)(n.strong,{children:"false"}),");"]}),"\n",(0,o.jsxs)(n.li,{children:["clear the contents of the column where your source values are\n(",(0,o.jsx)(n.strong,{children:"Edit cells"})," \u2192 ",(0,o.jsx)(n.strong,{children:"Common transformations"})," \u2192 ",(0,o.jsx)(n.strong,{children:"To null"}),")."]}),"\n"]}),"\n",(0,o.jsx)(n.p,{children:"You can now construct your schema as usual - null values will be ignored\nwhen generating the statements."})]})}function h(e={}){const{wrapper:n}={...(0,s.R)(),...e.components};return n?(0,o.jsx)(n,{...e,children:(0,o.jsx)(d,{...e})}):d(e)}},28453(e,n,t){t.d(n,{R:()=>i,x:()=>r});var a=t(96540);const o={},s=a.createContext(o);function i(e){const n=a.useContext(s);return a.useMemo((function(){return"function"==typeof e?e(n):{...n,...e}}),[n,e])}function r(e){let n;return n=e.disableParentContext?"function"==typeof e.components?e.components(o):e.components||o:i(e.components),a.createElement(s.Provider,{value:n},e.children)}}}]);
Line numbers count LF bytes from the start of the resource, as the search results do. Vendor segments are library code the classifier recognised; they are stored but not indexed. Bytes are shown as Latin1 characters, one per byte.